{"benchmarks": [{"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "视觉生成", "Visual Generation", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A-Bench is a benchmark designed to diagnose whether LMMs are masters at evaluating AIGIs. 2,864 AIGIs from 16 text-to-image models are sampled, each paired with question-answers annotated by human experts, and tested across 18 leading LMMs. A-Bench是一个旨在诊断 LMMs 是否擅长评估 AIGIs 的基准，从 16 个文本到图像模型中采样了 2,864 个 AIGIs，每个都与由人类专家标注的问题-答案配对。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1580", "languages": [], "modality": "multimodal", "name": "A-Bench", "openness": "unknown", "publisher": "SJTU, NTU.", "released": "2024-06-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1580-a-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/A-Bench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "知识", "Knowledge", "VQA", "多模态模型", "VLM", "逻辑推理", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A-OKVQA assesses commonsense reasoning abilities. It is a crowdsourced dataset composed of a diverse set of about 25K questions requiring a broad base of commonsense and world knowledge to answer. A-OKVQA用于评估多模态大模型的常识及推理能力，由25K个不同的问题组成，需要对图像中描述的场景进行某种形式的常识性推理来回答。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1367", "languages": [], "modality": "multimodal", "name": "A-OKVQA", "openness": "unknown", "publisher": "Allen Institute for AI", "released": "2022-06-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1367-a-okvqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/A-OKVQA", "unit": null}, {"aliases": [], "categories": ["agentic", "business", "reasoning"], "collected_at": "2026-08-25T10:41:06Z", "description": "Quantitative analysis on spreadsheets & documents", "evidence_summary": {"document_count": 1, "model_count": 30, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:aa-analystagent:a6340098-d7ae-462d-b372-0a0a67fc44b4", "reported_at": "2025-10-15", "source_url": "https://artificialanalysis.ai/evaluations/aa-analyst-agent"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:aa-analystagent", "languages": [], "modality": null, "name": "AA-AnalystAgent", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 30, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.0, "display_multiplier": 100, "model_count": 30, "model_count_basis": "source_model_id", "numeric_count": 30, "raw_max": 0.6, "raw_min": 0.0125, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:aa-analystagent:b2331108-72ed-415a-82d1-188633875bbc", "reported_date": "2026-08-13", "source_url": "https://artificialanalysis.ai/evaluations/aa-analyst-agent"}, "unit": null}, "slug": "artificial-analysis-aa-analystagent", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/aa-analyst-agent", "unit": null}, {"aliases": [], "categories": ["agentic", "business"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic knowledge work, Elo", "evidence_summary": {"document_count": 1, "model_count": 65, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:aa-briefcase:922c69c7-9037-43c6-8bcf-a1c555e7f3eb", "reported_at": "2025-04-05", "source_url": "https://artificialanalysis.ai/evaluations/aa-briefcase"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:aa-briefcase", "languages": [], "modality": null, "name": "AA-Briefcase", "openness": "unknown", "publisher": null, "released": "2026-06-18", "released_reference": {"basis": "release_announcement", "note": "The publisher launches AA-Briefcase and its four held-out knowledge-work scenarios on June 18.", "source_key": "artificial-analysis:aa-briefcase", "source_url": "https://artificialanalysis.ai/articles/aa-briefcase"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 65, "score_direction": "higher_is_better", "score_summary": {"display_max": 1710.26, "display_multiplier": 1, "model_count": 65, "model_count_basis": "source_model_id", "numeric_count": 65, "raw_max": 1710.26, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:aa-briefcase:b8fc61f7-5e9a-49e6-8547-6ac56db24627", "reported_date": "2026-07-24", "source_url": "https://artificialanalysis.ai/evaluations/aa-briefcase"}, "unit": null}, "slug": "artificial-analysis-aa-briefcase", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/aa-briefcase", "unit": null}, {"aliases": [], "categories": ["productivity", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AA-Briefcase is an Artificial Analysis evaluation of AI systems on professional knowledge-work tasks, reported as an Elo score.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aa-briefcase:kimi-k3", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aa-briefcase", "languages": [], "modality": "text", "name": "AA-Briefcase", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 1577.0, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 1577.0, "raw_min": 917.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aa-briefcase:grok-4.6", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500"}, "unit": null}, "slug": "llm-stats-aa-briefcase", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "No official academic documentation found for this benchmark. Extensive research through ArXiv, IEEE/ACL/NeurIPS papers, and university research sites yielded no peer-reviewed sources for an 'aa-index' benchmark. This entry requires verification from official academic sources.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aa-index:glm-4.5", "reported_at": "2025-07-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aa-index", "languages": [], "modality": "text", "name": "AA-Index", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.7, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.677, "raw_min": 0.61, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aa-index:glm-4.5", "reported_date": "2025-07-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500"}, "unit": null}, "slug": "llm-stats-aa-index", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "long-context", "reasoning"], "collected_at": "2026-08-25T10:41:06Z", "description": "Long context reasoning", "evidence_summary": {"document_count": 1, "model_count": 510, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:aa-lcr:217b34ec-5920-4fc1-8886-6a70a324837d", "reported_at": "2023-09-27", "source_url": "https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:aa-lcr", "languages": [], "modality": null, "name": "AA-LCR", "openness": "unknown", "publisher": null, "released": "2025-08-05", "released_reference": {"basis": "release_announcement", "note": "Artificial Analysis announces the release of this long-context reasoning benchmark, before its later inclusion in Index v2.2.", "source_key": "artificial-analysis:aa-lcr", "source_url": "https://artificialanalysis.ai/articles/announcing-aa-lcr"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 510, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.3333333333333, "display_multiplier": 100, "model_count": 510, "model_count_basis": "source_model_id", "numeric_count": 510, "raw_max": 0.833333333333333, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:aa-lcr:04ee6719-0327-463b-a1a1-70a6a78254f9", "reported_date": "2026-08-05", "source_url": "https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning"}, "unit": null}, "slug": "artificial-analysis-aa-lcr", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Agent Arena Long Context Reasoning benchmark", "evidence_summary": {"document_count": 1, "model_count": 18, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aa-lcr:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aa-lcr", "languages": [], "modality": "text", "name": "AA-LCR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 18, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.0, "display_multiplier": 100, "model_count": 18, "model_count_basis": "source_model_id", "numeric_count": 18, "raw_max": 0.8, "raw_min": 0.047, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aa-lcr:muse-glimmer-30b", "reported_date": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500"}, "unit": null}, "slug": "llm-stats-aa-lcr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500", "unit": null}, {"aliases": ["AA-LCR"], "categories": ["long_context"], "collected_at": null, "description": "Run by Artificial Analysis rather than the vendor. Third-party execution is the point, but it also means the vendor did not control the setup.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000aa_lcr\u0000qwen3_5_model_card\u0000aa_lcr\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000aa_lcr\u0000qwen3_5_model_card\u0000aa_lcr\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:aa_lcr", "languages": [], "modality": null, "name": "AA-LCR", "openness": "unknown", "publisher": null, "released": "2025-09-16", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:aa_lcr", "source_url": "https://artificialanalysis.ai/evaluations/aa-lcr"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.7, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 74.7, "raw_min": 68.7, "source_reference": {"obs_id": "curated\u0000aa_lcr\u0000moonshot_kimi_k3_model_card\u0000aa_lcr\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000aa_lcr\u0000moonshot_kimi_k3_model_card\u0000aa_lcr\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "aa_lcr", "source": "model_reports", "source_url": "https://artificialanalysis.ai/evaluations/aa-lcr", "unit": "percent"}, {"aliases": [], "categories": ["intelligence-index", "knowledge"], "collected_at": "2026-08-25T10:41:06Z", "description": "Knowledge", "evidence_summary": {"document_count": 1, "model_count": 489, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:aa-omniscience-accuracy:6a7c0e25-1dcb-4b15-8495-a8536a9da051", "reported_at": "2023-03-14", "source_url": "https://artificialanalysis.ai/evaluations/omniscience"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:aa-omniscience-accuracy", "languages": [], "modality": null, "name": "AA-Omniscience Accuracy", "openness": "unknown", "publisher": null, "released": "2025-11-16", "released_reference": {"basis": "release_announcement", "note": "The Omniscience launch introduces the factual-recall and hallucination evaluation; this source record retains its accuracy metric.", "source_key": "artificial-analysis:aa-omniscience-accuracy", "source_url": "https://artificialanalysis.ai/articles/aa-omniscience-knowledge-hallucination-benchmark"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 489, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.35, "display_multiplier": 100, "model_count": 489, "model_count_basis": "source_model_id", "numeric_count": 489, "raw_max": 0.6535, "raw_min": 0.001, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:aa-omniscience-accuracy:cd55210d-358e-4df1-ba9c-9acb5f186cc9", "reported_date": "2026-06-09", "source_url": "https://artificialanalysis.ai/evaluations/omniscience"}, "unit": null}, "slug": "artificial-analysis-aa-omniscience-accuracy", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/omniscience", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "knowledge"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AA-Omniscience Index is Artificial Analysis's knowledge-reliability metric. It rewards correct answers, penalizes hallucinations, and does not penalize abstention. Scores range from -100 to 100, where 0 means as many correct as incorrect answers.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aa-omniscience-index:grok-4.5", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aa-omniscience-index", "languages": [], "modality": "text", "name": "AA-Omniscience Index", "openness": "unknown", "publisher": null, "released": "2025-11-16", "released_reference": {"basis": "release_announcement", "note": "The source explicitly describes Artificial Analysis's knowledge-reliability index, introduced in this launch. Its signed index scale remains distinct from accuracy and non-hallucination rate.", "source_key": "llm-stats:aa-omniscience-index", "source_url": "https://artificialanalysis.ai/articles/aa-omniscience-knowledge-hallucination-benchmark"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 126.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 126.0, "raw_min": -29.5, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aa-omniscience-index:grok-4.5", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500"}, "unit": null}, "slug": "llm-stats-aa-omniscience-index", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "knowledge", "faithfulness"], "collected_at": "2026-08-25T10:41:06Z", "description": "1 - hallucination rate", "evidence_summary": {"document_count": 1, "model_count": 489, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:aa-omniscience-non-hallucination:6a7c0e25-1dcb-4b15-8495-a8536a9da051", "reported_at": "2023-03-14", "source_url": "https://artificialanalysis.ai/evaluations/omniscience"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:aa-omniscience-non-hallucination", "languages": [], "modality": null, "name": "AA-Omniscience Non-Hallucination Rate", "openness": "unknown", "publisher": null, "released": "2025-11-16", "released_reference": {"basis": "release_announcement", "note": "The Omniscience launch introduces the factual-recall and hallucination evaluation; this source record retains its non-hallucination metric.", "source_key": "artificial-analysis:aa-omniscience-non-hallucination", "source_url": "https://artificialanalysis.ai/articles/aa-omniscience-knowledge-hallucination-benchmark"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 489, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.09909909909909, "display_multiplier": 100, "model_count": 489, "model_count_basis": "source_model_id", "numeric_count": 489, "raw_max": 0.990990990990991, "raw_min": 0.015260480617435568, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:aa-omniscience-non-hallucination:d306cccd-0085-4b2f-8aa0-ffcdbb434695", "reported_date": "2026-05-25", "source_url": "https://artificialanalysis.ai/evaluations/omniscience"}, "unit": null}, "slug": "artificial-analysis-aa-omniscience-non-hallucination", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/omniscience", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "NAACL 2024", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ABSPYRAMID is a unified entailment graph of 221K textual descriptions of abstraction knowledge. ABSPYRAMID collects abstract knowledge for three components\nof diverse events to comprehensively evaluate the abstraction ability of language\nmodels in the open domain. ABSPYRAMID 是包含 221,000 条文本描述的抽象知识,收集了多种事件的三个组成部分的抽象知识，以全面评估语言模型在开放域中的抽象能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1148", "languages": [], "modality": null, "name": "AbsPyramid", "openness": "unknown", "publisher": "Tencent AI Lab", "released": "2024-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1148-abspyramid", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AbsPyramid", "unit": null}, {"aliases": [], "categories": ["reasoning", "finance", "general", "healthcare", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ACEBench is a comprehensive benchmark for evaluating Large Language Models' tool usage capabilities across three primary evaluation types: Normal (basic tool usage scenarios), Special (tool usage with ambiguous or incomplete instructions), and Agent (multi-agent interactions simulating real-world dialogues). The benchmark covers 4,538 APIs across 8 major domains and 68 sub-domains including technology, finance, entertainment, society, health, culture, and environment, supporting both English and Chinese languages.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:acebench:kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:acebench", "languages": [], "modality": "text", "name": "ACEBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.765, "raw_min": 0.765, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:acebench:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500"}, "unit": null}, "slug": "llm-stats-acebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "NeurIPS 2024", "多模态模型", "VLM", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ActionAtlas is a multiple-choice video question answering benchmark, including 934 videos showcasing 580 unique actions across 56 sports, with a total of 1896 actions within choices. ActionAtlas是一个多项选择视频问答基准测试，包括934个视频，展示了56项运动中的580个独特动作，选项共包含1896个动作。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1319", "languages": [], "modality": "multimodal", "name": "ActionAtlas", "openness": "unknown", "publisher": "University of Washington", "released": "2024-10-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1319-actionatlas", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ActionAtlas", "unit": null}, {"aliases": [], "categories": ["video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A large-scale video benchmark for human activity understanding. Provides samples from 203 activity classes with an average of 137 untrimmed videos per class and 1.41 activity instances per video, for a total of 849 video hours. The benchmark covers a wide range of complex human activities that are of interest to people in their daily living and can be used to compare algorithms for three scenarios: untrimmed video classification, trimmed activity classification, and activity detection.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:activitynet:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:activitynet", "languages": [], "modality": "video", "name": "ActivityNet", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.619, "raw_min": 0.619, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:activitynet:gpt-4o-2024-08-06", "reported_date": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500"}, "unit": null}, "slug": "llm-stats-activitynet", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "NAACL 2024", "大语言模型", "LLM", "长上下文", "Long Context", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Ada-LEval is a length-adaptable benchmark for evaluating the long-context understanding\nof LLMs. Ada-LEval includes two challenging subsets, TSort and BestAnswer, which enable\na more reliable evaluation of LLMs’ long context capabilities. Ada-LEval 用于评估大型语言模型（LLMs）对长上下文的理解能力。Ada-LEval 包含两个具有挑战性的子集，TSort 和 BestAnswer，能够更可靠地评估 LLMs 的长上下文能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1155", "languages": [], "modality": null, "name": "Ada-LEval", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1155-ada-leval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Ada-LEval", "unit": null}, {"aliases": [], "categories": ["reasoning", "instruction_following", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AdvancedIF is a rubric-based benchmark measuring complex, multi-turn, and system-prompted instruction following ability, scored with a calibrated LLM judge against per-instruction rubrics.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:advancedif:mai-code-1-flash", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:advancedif", "languages": [], "modality": "text", "name": "AdvancedIF", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.85, "raw_min": 0.714, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:advancedif:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500"}, "unit": null}, "slug": "llm-stats-advancedif", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500", "unit": null}, {"aliases": [], "categories": ["科学", "Science", "学科", "Examination", "知识", "Knowledge", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AECBench is an open-source benchmark for evaluating LLMs in architecture, engineering, and construction (AEC), covering 23 tasks and about 4,800 samples across five cognitive levels. AECBench 是面向建筑、工程与施工（AEC）领域的大语言模型评测基准，覆盖 5 个认知层级、23 类任务和约 4,800 个样本，用于评估模型在知识记忆、理解、推理、计算与应用方面的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2452", "languages": ["Chinese"], "modality": null, "name": "AECBench", "openness": "restricted", "publisher": "华东建筑设计研究院有限公司、同济大学", "released": "2026-04-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2452-aecbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AECBench", "unit": null}, {"aliases": [], "categories": [], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:community:5f95f778-c521-43fa-b80e-6a55465601e3", "languages": [], "modality": null, "name": "ael_gate_benchmark_cases_template", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-community-5f95f778-c521-43fa-b80e-6a55465601e3", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A5f95f778-c521-43fa-b80e-6a55465601e3?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "coding"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AetherCode is a competitive-programming benchmark of olympiad-level algorithmic coding problems.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aethercode:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aethercode", "languages": [], "modality": "text", "name": "AetherCode", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.9, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.679, "raw_min": 0.658, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aethercode:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500"}, "unit": null}, "slug": "llm-stats-aethercode", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AFQMC is an Ant Financial chinese semantic similarity task, which requires to judge whether two sentences have the same meaning or not. AFQMC一个蚂蚁金服中文语义相似度任务，要求判断两个句子是否具有相同的语义。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:506", "languages": ["Chinese"], "modality": null, "name": "AFQMC", "openness": "unknown", "publisher": null, "released": "2020-04-13", "released_reference": {"basis": "paper_first_version", "note": "CLUE's first paper version introduces the Ant Financial semantic-similarity task in section 4.2 and reports its evaluation.", "source_key": "opencompass:506", "source_url": "https://arxiv.org/abs/2004.05986v1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-506-afqmc", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AFQMC", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Agent Startup Bench measures AI agents on high-economic-value, startup-style tasks that require autonomous planning and execution to deliver practical, verifiable results.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:agent-startup-bench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:agent-startup-bench", "languages": [], "modality": "text", "name": "Agent Startup Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.688, "raw_min": 0.54, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:agent-startup-bench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-agent-startup-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "NeurIPS 2024", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AgentBoard is tailored to analytical evaluation of LLM agents. It offers a fine-grained progress rate metric that captures incremental advancements as well as a comprehensive evaluation toolkit that features easy assessment of agents for multi-faceted analysis through interactive visualization. AgentBoard专用于LLM Agent的分析评估，它提供了一个精细指标用于捕获增量进步，以及一个全面的评估工具包，能基于交互式可视化评估进行多方面分析。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1242", "languages": [], "modality": null, "name": "AgentBoard", "openness": "restricted", "publisher": "The University of Hong Kong", "released": "2024-06-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1242-agentboard", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentBoard", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "智能体", "Agent", "任务执行", "Task Execution", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AgentHarm tests the robustness of LLMs to jailbreak attacks. It includes a diverse set of 110 explicitly malicious agent tasks (440 with augmentations), covering 11 harm categories including fraud, cybercrime, and harassment. AgentHarm用于评估LLM智能体对越狱攻击的鲁棒性，包括110套恶意智能体任务（其中有440个强化任务），涵盖欺诈、网络犯罪和骚扰等11个危害类别。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1351", "languages": [], "modality": null, "name": "AgentHarm", "openness": "restricted", "publisher": "Gray Swan AI", "released": "2024-10-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1351-agentharm", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentHarm", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "安全", "Safety", "智能体", "Agent", "任务执行", "Task Execution", "安全对齐", "Safety Alignment", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "移动端 GUI Agent 通过与设备环境的交互来完成任务，在完成任务的过程中会遇到一些未知或不可信的信息来源，这些信息可能含有攻击性的内容，致使 Agent 无法正常完成任务，甚至对用户的隐私和财产带来危害。本评测集兼具动态执行环境和静态评测数据集，旨在为移动端 GUI Agent 提供一个仿真度高的模拟环境，以评估其在真实场景下执行的行为和安全性。 移动端 GUI Agent 通过与设备环境的交互来完成任务，在完成任务的过程中会遇到一些未知或不可信的信息来源，这些信息可能含有攻击性的内容，致使 Agent 无法正常完成任务，甚至对用户的隐私和财产带来危害。本评测集兼具动态执行环境和静态评测数据集，旨在为移动端 GUI Agent 提供一个仿真度高的模拟环境，以评估其在真实场景下执行的行为和安全性。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2061", "languages": [], "modality": "multimodal", "name": "AgentHazard", "openness": "unknown", "publisher": "Institute for AI Industry Research, Tsinghua University", "released": "2025-07-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2061-agenthazard", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentHazard", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AgentRewardBench, the first benchmark to assess the effectiveness of LLM judges for evaluating web agents. AgentRewardBench contains 1302 trajectories across 5 benchmarks and 4 LLMs. AgentRewardBench 是首个用于评估大型语言模型（LLM）评判者评估网络代理有效性的基准测试。AgentRewardBench 包含来自 5 个基准测试和 4 个大型语言模型的 1302 条轨迹。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1778", "languages": [], "modality": null, "name": "AgentRewardBench", "openness": "restricted", "publisher": "McGill University，Mila Quebec AI Institute，etc.", "released": "2025-04-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1778-agentrewardbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentRewardBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Agents' Last Exam is a challenging benchmark for AI agents on hard, long-horizon tasks that test sustained reasoning, planning, and tool use, reported with and without tool access.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:agents-last-exam:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:agents-last-exam", "languages": [], "modality": "text", "name": "Agents' Last Exam", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 52.7, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.527, "raw_min": 0.252, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:agents-last-exam:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500"}, "unit": null}, "slug": "llm-stats-agents-last-exam", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500", "unit": null}, {"aliases": ["Agents' Last Exam", "ALE", "Agent's Last Exam", "ALE-CLI"], "categories": ["coding_agent"], "collected_at": null, "description": "Multi-step agentic benchmark using Claude Code harness. Tool Search disabled. Scores depend on harness, reasoning effort, context length, and timeout settings.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000agents_last_exam\u0000zai_glm_5_3_flash_model_card\u0000agents_last_exam\u0000official evaluation protocol, Claude Code harness, reasoning effort=max, 1M context, 64K max output, Tool Search disabled, scored by official ALE evaluators\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000agents_last_exam\u0000zai_glm_5_3_flash_model_card\u0000agents_last_exam\u0000official evaluation protocol, Claude Code harness, reasoning effort=max, 1M context, 64K max output, Tool Search disabled, scored by official ALE evaluators\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:agents_last_exam", "languages": [], "modality": null, "name": "Agents' Last Exam", "openness": "unknown", "publisher": null, "released": "2025-01-23", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:agents_last_exam", "source_url": "https://lastexam.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 26.3, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 26.3, "raw_min": 22.8, "source_reference": {"obs_id": "curated\u0000agents_last_exam\u0000zai_glm_5_3_flash_model_card\u0000agents_last_exam\u0000official evaluation protocol, Claude Code harness, reasoning effort=max, 1M context, 64K max output, Tool Search disabled, scored by official ALE evaluators\u0000GLM-5.3-Flash", "observation_id": "curated\u0000agents_last_exam\u0000zai_glm_5_3_flash_model_card\u0000agents_last_exam\u0000official evaluation protocol, Claude Code harness, reasoning effort=max, 1M context, 64K max output, Tool Search disabled, scored by official ALE evaluators\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "agents_last_exam", "source": "model_reports", "source_url": "https://lastexam.ai/", "unit": "percent"}, {"aliases": [], "categories": ["legal", "math", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A human-centric benchmark for evaluating foundation models on standardized exams including college entrance exams (Gaokao, SAT), law school admission tests (LSAT), math competitions, lawyer qualification tests, and civil service exams. Contains 20 tasks (18 multiple-choice, 2 cloze) designed to assess understanding, knowledge, reasoning, and calculation abilities in real-world academic and professional contexts.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:agieval:gemma-2-27b-it", "reported_at": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:agieval", "languages": [], "modality": "text", "name": "AGIEval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.8, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.658, "raw_min": 0.285, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:agieval:mistral-small-24b-base-2501", "reported_date": "2025-01-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500"}, "unit": null}, "slug": "llm-stats-agieval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500", "unit": null}, {"aliases": ["AGIEval"], "categories": ["knowledge"], "collected_at": null, "description": "Human-exam derived; overlaps heavily with MMLU-style coverage.", "evidence_summary": {"document_count": 4, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:agieval", "languages": [], "modality": null, "name": "AGIEval", "openness": "unknown", "publisher": null, "released": "2023-04-13", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:agieval", "source_url": "https://github.com/ruixiangcui/AGIEval"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "agieval", "source": "model_reports", "source_url": "https://github.com/ruixiangcui/AGIEval", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AGIEval is a human-centric benchmark specifically designed to evaluate the general abilities of foundation models in tasks pertinent to human cognition and problem-solving. This benchmark is derived from 20 official, public, and high-standard admission and qualification exams intended for general human test-takers, such as general college admission tests (e.g., Chinese College Entrance Exam (Gaokao) and American SAT), law school admission tests, math competitions, lawyer qualification tests, and national civil service exams. AGIEval是一个以人为中心的基准测试，专门设计用于评估基础模型在涉及人类认知和问题解决的任务中的一般能力。该基准测试源自20个官方、公开和高标准的入学和资格考试，例如普通大学入学考试（例如中国高考和美国SAT）、法学院入学考试、数学竞赛、律师资格考试以及国家公务员考试", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:497", "languages": ["Chinese"], "modality": null, "name": "AGIEval", "openness": "unknown", "publisher": null, "released": "2023-09-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-497-agieval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AGIEval", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "农业", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A Comprehensive Agricultural Multimodal Understanding and Reasoning Benchmark 农业综合多模态理解和推理基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1753", "languages": [], "modality": "multimodal", "name": "AgMMU", "openness": "unknown", "publisher": "Rice University，etc.", "released": "2025-04-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1753-agmmu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgMMU", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A dataset of 7,787 genuine grade-school level, multiple-choice science questions assembled to encourage research in advanced question-answering. The dataset is partitioned into a Challenge Set and Easy Set, where the Challenge Set contains only questions answered incorrectly by both retrieval-based and word co-occurrence algorithms. Covers multiple scientific domains including biology, physics, earth science, and chemistry, requiring scientific reasoning, causal understanding, and conceptual knowledge beyond simple fact retrieval. Includes a supporting corpus of over 14 million science sentences.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ai2-reasoning-challenge-(arc):gpt-4-0613", "reported_at": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ai2-reasoning-challenge-(arc)", "languages": [], "modality": "text", "name": "AI2 Reasoning Challenge (ARC)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.963, "raw_min": 0.963, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ai2-reasoning-challenge-(arc):gpt-4-0613", "reported_date": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500"}, "unit": null}, "slug": "llm-stats-ai2-reasoning-challenge-arc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AI2D is a dataset of 4,903 illustrative diagrams from grade school natural sciences (such as food webs, human physiology, and life cycles) with over 15,000 multiple choice questions and answers. The benchmark evaluates diagram understanding and visual reasoning capabilities, requiring models to interpret diagrammatic elements, relationships, and structure to answer questions about scientific concepts represented in visual form.", "evidence_summary": {"document_count": 1, "model_count": 33, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ai2d:grok-1.5v", "reported_at": "2024-04-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ai2d", "languages": [], "modality": "multimodal", "name": "AI2D", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 33, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.69999999999999, "display_multiplier": 100, "model_count": 33, "model_count_basis": "source_model_id", "numeric_count": 33, "raw_max": 0.947, "raw_min": 0.716, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ai2d:claude-3-5-sonnet-20241022", "reported_date": "2024-10-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500"}, "unit": null}, "slug": "llm-stats-ai2d", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "智能体", "Agent", "AI数据分析", "生产力工具", "数据可视化", "逻辑推理", "任务执行", "Task Execution", "数理能力", "Math", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AIDABench provides a professional benchmarking suite for AI office data analytics tools, focusing on the core data-processing needs in everyday office workflows, covering high-frequency and reproducible data-processing scenarios commonly seen in real business contexts. AIDABench致力于为AI办公数据分析工具提供专业评测基准，聚焦日常办公中的核心数据处理需求。评测输入以 Excel 文件为主，辅以少量DOC、 PDF、图片及无文件文本等形态，覆盖真实业务中高频、可复现的数据处理场景。核心考察能力包括：数据结构识别、数据清洗、条件筛选、分组聚合、描述统计、排序与排名。任务交付类型包括：问答（Q/A）、文件生成、数据可视化。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:2396", "languages": [], "modality": "multimodal", "name": "AIDABench", "openness": "unknown", "publisher": "商汤科技 & 上海人工智能实验室", "released": "2026-02-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2396-aidabench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIDABench", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Aider is a comprehensive code editing benchmark based on 133 practice exercises from Exercism's Python repository, designed to evaluate AI models' ability to translate natural language coding requests into executable code that passes unit tests. The benchmark measures end-to-end code editing capabilities, including GPT's ability to edit existing code and format code changes for automated saving to local files. The Aider Polyglot variant extends this evaluation across 225 challenging exercises spanning C++, Go, Java, JavaScript, Python, and Rust, making it a standard benchmark for assessing multilingual code editing performance in AI research.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aider:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aider", "languages": [], "modality": "text", "name": "Aider", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.2, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.722, "raw_min": 0.502, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aider:deepseek-v2.5", "reported_date": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500"}, "unit": null}, "slug": "llm-stats-aider", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500", "unit": null}, {"aliases": ["Aider Polyglot", "Aider"], "categories": ["coding"], "collected_at": null, "description": "Edit-format sensitive; whole-file and diff modes differ substantially.", "evidence_summary": {"document_count": 4, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000aider_polyglot\u0000deepseek_r1_report\u0000aider_polyglot\u0000Accuracy\u0000DeepSeek-R1", "reported_at": "2025-01-22", "source_url": "https://arxiv.org/abs/2501.12948"}, "first_score_reported_at": "2025-01-22", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000aider_polyglot\u0000deepseek_r1_report\u0000aider_polyglot\u0000Accuracy\u0000DeepSeek-R1", "reported_at": "2025-01-22", "source_url": "https://arxiv.org/abs/2501.12948"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:aider_polyglot", "languages": [], "modality": null, "name": "Aider Polyglot", "openness": "unknown", "publisher": null, "released": "2024-12-21", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:aider_polyglot", "source_url": "https://aider.chat/docs/leaderboards/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.2, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 82.2, "raw_min": 53.3, "source_reference": {"obs_id": "curated\u0000aider_polyglot\u0000google_gemini_2_5_report\u0000aider_polyglot\u0000pass rate, average of 3 trials\u0000Gemini 2.5 Pro", "observation_id": "curated\u0000aider_polyglot\u0000google_gemini_2_5_report\u0000aider_polyglot\u0000pass rate, average of 3 trials\u0000Gemini 2.5 Pro", "reported_at": "2025-06-17", "reported_date": "2025-06-17", "source_id": "google_gemini_2_5_report", "source_url": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf"}, "unit": "percent"}, "slug": "aider_polyglot", "source": "model_reports", "source_url": "https://aider.chat/docs/leaderboards/", "unit": "percent"}, {"aliases": [], "categories": ["general", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A coding benchmark that evaluates LLMs on 225 challenging Exercism programming exercises across C++, Go, Java, JavaScript, Python, and Rust. Models receive two attempts to solve each problem, with test error feedback provided after the first attempt if it fails. The benchmark measures both initial problem-solving ability and capacity to edit code based on error feedback, providing an end-to-end evaluation of code generation and editing capabilities across multiple programming languages.", "evidence_summary": {"document_count": 1, "model_count": 22, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aider-polyglot:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:aider-polyglot", "languages": [], "modality": "text", "name": "Aider-Polyglot", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 22, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 22, "model_count_basis": "source_model_id", "numeric_count": 22, "raw_max": 0.88, "raw_min": 0.098, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aider-polyglot:gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500"}, "unit": null}, "slug": "llm-stats-aider-polyglot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500", "unit": null}, {"aliases": [], "categories": ["general", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A challenging multi-language coding benchmark that evaluates models' code editing abilities across C++, Go, Java, JavaScript, Python, and Rust. Contains 225 of Exercism's most difficult programming problems, selected as problems that were solved by 3 or fewer out of 7 top coding models. The benchmark focuses on code editing tasks and measures both correctness of solutions and proper edit format usage. Designed to re-calibrate evaluation scales so top models score between 5-50%.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aider-polyglot-edit:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aider-polyglot-edit", "languages": [], "modality": "text", "name": "Aider-Polyglot Edit", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 79.7, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.797, "raw_min": 0.062, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aider-polyglot-edit:deepseek-v3", "reported_date": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500"}, "unit": null}, "slug": "llm-stats-aider-polyglot-edit", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "American Invitational Mathematics Examination (AIME) benchmark for evaluating mathematical reasoning capabilities of large language models. Contains 30 challenging mathematical problems from AIME 2024 competition that require multi-step reasoning and advanced mathematical insight. Each problem has an integer answer between 000-999.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aime:phi-4-mini-reasoning", "reported_at": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aime", "languages": [], "modality": "text", "name": "AIME", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.49999999999999, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.575, "raw_min": 0.373, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aime:phi-4-mini-reasoning", "reported_date": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500"}, "unit": null}, "slug": "llm-stats-aime", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500", "unit": null}, {"aliases": ["AIME", "AIME 2024", "AIME 2025", "AIME 2026"], "categories": ["math"], "collected_at": null, "description": "pass@k, majority vote and Python tool access each shift this by double digits.", "evidence_summary": {"document_count": 17, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000aime\u0000deepseek_v3_report\u0000aime_2024\u0000Pass@1\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000aime\u0000deepseek_v3_report\u0000aime_2024\u0000Pass@1\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:aime", "languages": [], "modality": null, "name": "AIME", "openness": "unknown", "publisher": null, "released": "2024-02-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:aime", "source_url": "https://maa.org/maa-invitational-competitions/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.2, "display_multiplier": 1, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 99.2, "raw_min": 39.2, "source_reference": {"obs_id": "curated\u0000aime\u0000zai_glm_5_2_model_card\u0000aime\u0000AIME 2026, temp 1.0, top_p 0.95, max gen 163840 tokens, GPT-5.5 (medium) judge\u0000GLM-5.2", "observation_id": "curated\u0000aime\u0000zai_glm_5_2_model_card\u0000aime\u0000AIME 2026, temp 1.0, top_p 0.95, max gen 163840 tokens, GPT-5.5 (medium) judge\u0000GLM-5.2", "reported_at": "2026-06-16", "reported_date": "2026-06-16", "source_id": "zai_glm_5_2_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5.2"}, "unit": "percent"}, "slug": "aime", "source": "model_reports", "source_url": "https://maa.org/maa-invitational-competitions/", "unit": "percent"}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "American Invitational Mathematics Examination 2024, consisting of 30 challenging mathematical reasoning problems from AIME I and AIME II competitions. Each problem requires an integer answer between 0-999 and tests advanced mathematical reasoning across algebra, geometry, combinatorics, and number theory. Used as a benchmark for evaluating mathematical reasoning capabilities in large language models at Olympiad-level difficulty.", "evidence_summary": {"document_count": 1, "model_count": 53, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aime-2024:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aime-2024", "languages": [], "modality": "text", "name": "AIME 2024", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 53, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.8, "display_multiplier": 100, "model_count": 53, "model_count_basis": "source_model_id", "numeric_count": 53, "raw_max": 0.958, "raw_min": 0.131, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aime-2024:grok-3-mini", "reported_date": "2025-02-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500"}, "unit": null}, "slug": "llm-stats-aime-2024", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "math"], "collected_at": "2026-08-25T10:41:06Z", "description": "Mathematical reasoning", "evidence_summary": {"document_count": 1, "model_count": 270, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:aime-2025:9f873c2f-2c2d-4ccb-9e1b-71bf61b052be", "reported_at": "2024-02-26", "source_url": "https://artificialanalysis.ai/evaluations/aime-2025"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:aime-2025", "languages": [], "modality": null, "name": "AIME 2025", "openness": "unknown", "publisher": null, "released": "2025-02-12", "released_reference": {"basis": "dataset_published", "note": "This source evaluates all 30 AIME I and II problems. MAA dates AIME I to February 6 and AIME II to February 12; the latter completes the 2025 problem set.", "source_key": "artificial-analysis:aime-2025", "source_url": "https://maa.org/news/aime-thresholds-are-available/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 270, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.0, "display_multiplier": 100, "model_count": 270, "model_count_basis": "source_model_id", "numeric_count": 270, "raw_max": 0.99, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:aime-2025:498862c3-f9ac-49d2-852f-16a02bb0c38f", "reported_date": "2025-12-11", "source_url": "https://artificialanalysis.ai/evaluations/aime-2025"}, "unit": null}, "slug": "artificial-analysis-aime-2025", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/aime-2025", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "All 30 problems from the 2025 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.", "evidence_summary": {"document_count": 1, "model_count": 115, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aime-2025:deepseek-v3.1", "reported_at": "2025-01-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:aime-2025", "languages": [], "modality": "text", "name": "AIME 2025", "openness": "unknown", "publisher": null, "released": "2025-02-12", "released_reference": {"basis": "dataset_published", "note": "The source explicitly evaluates all 30 AIME I and II problems. MAA dates the second exam to February 12, completing the 2025 problem set.", "source_key": "llm-stats:aime-2025", "source_url": "https://maa.org/news/aime-thresholds-are-available/"}, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 115, "score_direction": "higher_is_better", "score_summary": {"display_max": 100.0, "display_multiplier": 100, "model_count": 115, "model_count_basis": "source_model_id", "numeric_count": 115, "raw_max": 1.0, "raw_min": 0.067, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aime-2025:gemini-3-pro-preview", "reported_date": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500"}, "unit": null}, "slug": "llm-stats-aime-2025", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "All 30 problems from the 2026 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.", "evidence_summary": {"document_count": 1, "model_count": 21, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aime-2026:seed-2.0-lite", "reported_at": "2026-02-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:aime-2026", "languages": [], "modality": "text", "name": "AIME 2026", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 21, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.2, "display_multiplier": 100, "model_count": 21, "model_count_basis": "source_model_id", "numeric_count": 21, "raw_max": 0.992, "raw_min": 0.375, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aime-2026:glm-5.2", "reported_date": "2026-06-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500"}, "unit": null}, "slug": "llm-stats-aime-2026", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500", "unit": null}, {"aliases": [], "categories": ["safety"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AIR-Bench 2024 is a safety benchmark grounded in risk categories derived from government regulations and company policies. It evaluates policy-grounded refusal across a broad regulatory and policy-derived harm taxonomy, using category-specific LLM-judge prompts that reward safe engagement rather than only penalizing unsafe responses.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:air-bench:mai-thinking-1", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:air-bench", "languages": [], "modality": "text", "name": "AIR-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.88, "raw_min": 0.88, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:air-bench:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-air-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "ACL 2024", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AIR-Bench is the first benchmark designed to evaluate the\nability of LALMs to understand various types of audio signals (including human speech, natural sounds, and music), and furthermore, to interact with humans in the textual format. AIR-Bench 是第一个旨在评估 LALMs 理解各种音频信号（包括人类语言、自然声音和音乐）能力的基准，并进一步评估其以文本形式与人类互动的能力。AIR-Bench 包括两个维度：基础基准和聊天基准。前者由19个任务组成，包含约19,000个单选题，旨在检查LALMs的基本单任务能力。后者包含2,000个开放式问答数据实例，直接评估模型对复杂音频的理解及其遵循指令的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1069", "languages": [], "modality": null, "name": "AIR-Bench", "openness": "unknown", "publisher": "Zhejiang University, Alibaba Group", "released": "2024-02-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1069-air-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIR-Bench", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AIR-Bench 2024, the first policy-aligned AI safety benchmark, structures 8 government regulations and 16 corporate policies into four security tiers, with 5,694 diverse prompts spanning these categories. AIR-Bench 2024是首个与新兴政府法规和企业政策相一致的 AI 安全基准， 将 8 项政府法规和 16 项企业政策分解为四级安全分类，涵盖了这些类别的 5,694 个多样化的提示。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1557", "languages": [], "modality": null, "name": "AIRBench-2024", "openness": "restricted", "publisher": "Virtue AI,  Virginia Tech, University of California, Los Angeles, etc.", "released": "2024-08-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1557-airbench-2024", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIRBench-2024", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AIRTBench is a benchmark designed to evaluate large language models (LLMs) on their autonomous AI red teaming capabilities. AIRTBench 是一个专为评估大型语言模型（LLM）在“红队”安全任务中的自主攻击能力而设计的评测基准。本基准包含 70 个黑盒 CTF（夺旗赛）挑战，模拟真实 AI/ML 系统漏洞环境，要求模型独立编写 Python 代码进行漏洞发现、利用与夺旗操作，体现其计划、推理与系统操控等综合能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1995", "languages": [], "modality": null, "name": "AIRTBench", "openness": "open", "publisher": "dreadnode", "released": "2025-06-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1995-airtbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIRTBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Android-In-The-Zoo (AitZ) benchmark for evaluating autonomous GUI agents on smartphones. Contains 18,643 screen-action pairs with chain-of-action-thought annotations spanning over 70 Android apps. Designed to connect perception (screen layouts and UI elements) with cognition (action decision-making) for natural language-triggered smartphone task completion.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:aitz-em:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:aitz-em", "languages": [], "modality": "multimodal", "name": "AITZ_EM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.832, "raw_min": 0.819, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:aitz-em:qwen2.5-vl-72b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500"}, "unit": null}, "slug": "llm-stats-aitz-em", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ALE-Bench is a benchmark for evaluating AI systems on score-based algorithmic programming contests. ALE-Bench 是一个用于评估 AI 系统在基于分数的算法编程竞赛中的基准测试。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1969", "languages": ["English", "Japanese"], "modality": null, "name": "ALE-Bench", "openness": "restricted", "publisher": "SakanaAI, Japan , The University of Tokyo, Japan , AtCoder, etc.", "released": "2025-06-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1969-ale-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ALE-Bench", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "roleplay", "language", "general", "creativity", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AlignBench is a comprehensive multi-dimensional benchmark for evaluating Chinese alignment of Large Language Models. It contains 8 main categories: Fundamental Language Ability, Advanced Chinese Understanding, Open-ended Questions, Writing Ability, Logical Reasoning, Mathematics, Task-oriented Role Play, and Professional Knowledge. The benchmark includes 683 real-scenario rooted queries with human-verified references and uses a rule-calibrated multi-dimensional LLM-as-Judge approach with Chain-of-Thought for evaluation.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:alignbench:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:alignbench", "languages": [], "modality": "text", "name": "AlignBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.6, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.816, "raw_min": 0.721, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:alignbench:qwen-2.5-72b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500"}, "unit": null}, "slug": "llm-stats-alignbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "ACL 2024", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ALIGNBENCH is a comprehensive multidimensional benchmark for evaluating LLMs’ alignment in Chinese. We tailor a humanin-the-loop data curation pipeline, containing 8 main categories, 683 real-scenario rooted queries and corresponding human verified references. AlignBench 是一个用于评估中文大语言模型对齐性能的全面、多维度的评测基准。AlignBench 构建了人类参与的数据构建流程，来保证评测数据的动态更新。AlignBench 采用多维度、规则校准的模型评价方法（LLM-as-Judge），并且结合思维链（Chain-of-Thought）生成对模型回复的多维度分析和最终的综合评分，增强了评测的高可靠性和可解释性。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1075", "languages": ["Chinese"], "modality": null, "name": "AlignBench", "openness": "unknown", "publisher": "THUDM", "released": "2024-08-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1075-alignbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AlignBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "creativity", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AlpacaEval 2.0 is a length-controlled automatic evaluator for instruction-following language models that uses GPT-4 Turbo to assess model responses against a baseline. It evaluates models on 805 diverse instruction-following tasks including creative writing, classification, programming, and general knowledge questions. The benchmark achieves 0.98 Spearman correlation with ChatBot Arena while being fast (< 3 minutes) and affordable (< $10 in OpenAI credits). It addresses length bias in automatic evaluation through length-controlled win-rates and uses weighted scoring based on response quality.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:alpacaeval-2.0:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:alpacaeval-2.0", "languages": [], "modality": "text", "name": "AlpacaEval 2.0", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.68, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.6268, "raw_min": 0.3516, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:alpacaeval-2.0:granite-3.3-8b-base", "reported_date": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500"}, "unit": null}, "slug": "llm-stats-alpacaeval-2-0", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "大语言模型", "LLM", "代码工程", "Code", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AMBROSIA is a new benchmark for recognizing and interpreting ambiguous requests in text-to-SQL. It contains questions showcasing three different types of ambiguity (scope ambiguity, attachment ambiguity, and vagueness), their interpretations, and corresponding SQL queries. AMBROSIA是识别和解释text-to-SQL中歧义请求的新基准，其中包含三种不同类型的歧义（范围歧义、附件歧义和模糊性）问题、它们的解释和相应的SQL查询。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1280", "languages": [], "modality": null, "name": "AMBROSIA", "openness": "unknown", "publisher": "University of Edinburgh", "released": "2024-06-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1280-ambrosia", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AMBROSIA", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "American Mathematics Competition problems from the 2022-23 academic year, consisting of multiple-choice mathematics competition problems designed for high school students. These problems require advanced mathematical reasoning, problem-solving strategies, and mathematical knowledge covering topics like algebra, geometry, number theory, and combinatorics. The benchmark is derived from the official AMC competitions sponsored by the Mathematical Association of America.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:amc-2022-23:gemini-1.5-flash", "reported_at": "2024-05-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:amc-2022-23", "languages": [], "modality": "text", "name": "AMC_2022_23", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 52.0, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.52, "raw_min": 0.348, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:amc-2022-23:mistral-large-3-675B-instruct-2512-eagle", "reported_date": "2025-12-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500"}, "unit": null}, "slug": "llm-stats-amc-2022-23", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AMO Bench is an olympiad-level mathematics benchmark that evaluates advanced mathematical problem-solving and multi-step reasoning on competition-style problems.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:amo-bench:mai-code-1-flash", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:amo-bench", "languages": [], "modality": "text", "name": "AMO Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 40.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.4, "raw_min": 0.4, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:amo-bench:mai-code-1-flash", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-amo-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "创作", "Creation", "AMS", "Circuit", "EDA", "科学智能", "AI for Science", "图像理解", "Image Understanding", "语言生成", "Generation", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AMSbench is a benchmark of ~8000 questions to evaluate multi-modal LLMs on analog/mixed-signal circuit tasks like schematic recognition, analysis, and design. AMSbench 是一个包含约8000道题目的基准测试集，用于评估多模态大语言模型在模拟/混合信号电路任务中的表现，包括识图、分析与设计。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1966", "languages": [], "modality": "multimodal", "name": "AMSbench", "openness": "unknown", "publisher": "宁波东方理工大学", "released": "2025-06-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1966-amsbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AMSbench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Android device control benchmark using high exact match evaluation metric for assessing agent performance on mobile interface tasks", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:android-control-high-em:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:android-control-high-em", "languages": [], "modality": "multimodal", "name": "Android Control High_EM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 69.6, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.696, "raw_min": 0.601, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:android-control-high-em:qwen2.5-vl-32b", "reported_date": "2025-02-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500"}, "unit": null}, "slug": "llm-stats-android-control-high-em", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Android control benchmark evaluating autonomous agents on mobile device interaction tasks with low exact match scoring criteria", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:android-control-low-em:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:android-control-low-em", "languages": [], "modality": "multimodal", "name": "Android Control Low_EM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.937, "raw_min": 0.914, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:android-control-low-em:qwen2.5-vl-72b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500"}, "unit": null}, "slug": "llm-stats-android-control-low-em", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AndroidBench evaluates coding agents on Android application development tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:androidbench:qwen3.8-max", "reported_at": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:androidbench", "languages": [], "modality": "multimodal", "name": "AndroidBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.751, "raw_min": 0.751, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:androidbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500"}, "unit": null}, "slug": "llm-stats-androidbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AndroidWorld evaluates an agent's ability to operate in real Android GUI environments, completing multi-step tasks by perceiving screen content and executing touch/type actions.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:androidworld:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:androidworld", "languages": [], "modality": "multimodal", "name": "AndroidWorld", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.3, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.853, "raw_min": 0.703, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:androidworld:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500"}, "unit": null}, "slug": "llm-stats-androidworld", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AndroidWorld Success Rate (SR) benchmark - A dynamic benchmarking environment for autonomous agents operating on Android devices. Evaluates agents on 116 programmatic tasks across 20 real-world Android apps using multimodal inputs (screen screenshots, accessibility trees, and natural language instructions). Measures success rate of agents completing tasks like sending messages, creating calendar events, and navigating mobile interfaces. Published at ICLR 2025. Best current performance: 30.6% success rate (M3A agent) vs 80.0% human performance.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:androidworld-sr:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:androidworld-sr", "languages": [], "modality": "multimodal", "name": "AndroidWorld_SR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.1, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.711, "raw_min": 0.22, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:androidworld-sr:qwen3.5-35b-a3b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500"}, "unit": null}, "slug": "llm-stats-androidworld-sr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "Medical Imagery", "Aneurysm", "Computational Fluid Dynamics", "科学智能", "AI for Science", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Based on 427 real aneurysm geometries, we synthesized 10,660 3D shapes via controlled deformation to simulate aneurysm evolution. CFD computations were performed on each shape under eight steady-state mass flow conditions, generating a total of 85,280 blood flow dynamics data covering key parameters Based on 427 real aneurysm geometries, we synthesized 10,660 3D shapes via controlled deformation to simulate aneurysm evolution. CFD computations were performed on each shape under eight steady-state mass flow conditions, generating a total of 85,280 blood flow dynamics data covering key parameters", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1876", "languages": [], "modality": "multimodal", "name": "Aneumo", "openness": "restricted", "publisher": "Shanghai Academy of Artificial Intelligence for Science", "released": "2025-05-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1876-aneumo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Aneumo", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Apex is a challenging frontier reasoning benchmark testing advanced multi-step problem solving across difficult STEM and logical tasks.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:apex:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:apex", "languages": [], "modality": "text", "name": "Apex", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.848, "raw_min": 0.227, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:apex:nemotron-3-ultra-550b-a55b", "reported_date": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500"}, "unit": null}, "slug": "llm-stats-apex", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "APEX-Agents is a benchmark evaluating AI agents on long horizon professional tasks that require sustained reasoning, planning, and execution across complex multi-step workflows.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:apex-agents:gemini-3.1-pro-preview", "reported_at": "2026-02-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:apex-agents", "languages": [], "modality": "text", "name": "APEX-Agents", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.49999999999999, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.575, "raw_min": 0.256, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:apex-agents:grok-4.6", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500"}, "unit": null}, "slug": "llm-stats-apex-agents", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500", "unit": null}, {"aliases": ["APEX-Agents", "APEX", "Apex", "Apex Shortlist"], "categories": ["agent"], "collected_at": null, "description": "Expert-authored professional tasks with rubric grading; graders are LLMs.", "evidence_summary": {"document_count": 5, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000apex_agents\u0000google_gemini_3_1_pro_model_card\u0000apex_agents\u0000Thinking (High)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "first_score_reported_at": "2026-02-19", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000apex_agents\u0000google_gemini_3_1_pro_model_card\u0000apex_agents\u0000Thinking (High)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:apex_agents", "languages": [], "modality": null, "name": "APEX-Agents", "openness": "unknown", "publisher": null, "released": "2026-01-20", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:apex_agents", "source_url": "https://arxiv.org/abs/2601.14242"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 41.0, "display_multiplier": 1, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 41.0, "raw_min": 33.0, "source_reference": {"obs_id": "curated\u0000apex_agents\u0000moonshot_kimi_k3_model_card\u0000apex_agents\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000apex_agents\u0000moonshot_kimi_k3_model_card\u0000apex_agents\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "apex_agents", "source": "model_reports", "source_url": "https://arxiv.org/abs/2601.14242", "unit": "percent"}, {"aliases": [], "categories": ["agentic", "reasoning"], "collected_at": "2026-08-25T10:41:06Z", "description": "Long-horizon agentic tasks", "evidence_summary": {"document_count": 1, "model_count": 29, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:apex-agents-aa:36f73aaf-d38a-4b56-a2b3-d04d17186910", "reported_at": "2025-08-05", "source_url": "https://artificialanalysis.ai/evaluations/apex-agents-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:apex-agents-aa", "languages": [], "modality": null, "name": "APEX-Agents-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 29, "score_direction": "higher_is_better", "score_summary": {"display_max": 47.0501474926254, "display_multiplier": 100, "model_count": 29, "model_count_basis": "source_model_id", "numeric_count": 29, "raw_max": 0.470501474926254, "raw_min": 0.00737463126843658, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:apex-agents-aa:0097ebf5-124f-42f6-9463-33b00e711f03", "reported_date": "2026-05-19", "source_url": "https://artificialanalysis.ai/evaluations/apex-agents-aa"}, "unit": null}, "slug": "artificial-analysis-apex-agents-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/apex-agents-aa", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "APEX-SWE evaluates AI agents on software engineering tasks requiring multi-step coding, debugging, and verification.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:apex-swe:grok-4.6", "reported_at": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:apex-swe", "languages": [], "modality": "text", "name": "APEX-SWE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.39999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.564, "raw_min": 0.564, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:apex-swe:grok-4.6", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500"}, "unit": null}, "slug": "llm-stats-apex-swe", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive benchmark for tool-augmented LLMs that evaluates API planning, retrieval, and calling capabilities. Contains 314 tool-use dialogues with 753 API calls across 73 API tools, designed to assess how effectively LLMs can utilize external tools and overcome obstacles in tool leveraging.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:api-bank:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:api-bank", "languages": [], "modality": "text", "name": "API-Bank", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.92, "raw_min": 0.826, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:api-bank:llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500"}, "unit": null}, "slug": "llm-stats-api-bank", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "APPS is a benchmark for code generation. Unlike prior work in more restricted settings, our benchmark measures the ability of models to take an arbitrary natural language specification and generate satisfactory Python code. APPS 是一个代码生成评测基准，该评测基准测量模型根据任意自然语言规范生成令人满意的 Python 代码的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1093", "languages": [], "modality": null, "name": "APPS", "openness": "open", "publisher": "UC Berkeley", "released": "2021-11-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1093-apps", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/APPS", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "大语言模型", "LLM", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AQUA-RAT contains the algebraic word problems. The dataset consists of about 100,000 algebraic word problems with natural language rationales. AQUA-RAT 包含代数文字问题。该数据集由约 100,000 道带有自然语言推理的代数文字问题组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1116", "languages": [], "modality": null, "name": "AQUA-RAT", "openness": "open", "publisher": "DeepMind", "released": "2017-10-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1116-aqua-rat", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AQUA-RAT", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The Abstraction and Reasoning Corpus (ARC) is a benchmark designed to measure human-like general fluid intelligence through grid-based reasoning tasks. It consists of 800 tasks (400 training, 400 evaluation) where each task presents input-output grids that require understanding abstract patterns and transformations. Test-takers must produce exactly correct output grids for all test inputs in a task to solve it, with 3 trials allowed per test input. ARC aims to enable fair comparisons of general intelligence between AI systems and humans using priors designed to be as close as possible to innate human priors.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arc:gemini-2.5-flash-lite", "reported_at": "2025-06-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arc", "languages": [], "modality": "multimodal", "name": "Arc", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 2.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.025, "raw_min": 0.025, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arc:gemini-2.5-flash-lite", "reported_date": "2025-06-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500"}, "unit": null}, "slug": "llm-stats-arc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) is a benchmark designed to test general intelligence and abstract reasoning capabilities through visual grid-based transformation tasks. Each task consists of 2-5 demonstration pairs showing input grids transformed into output grids according to underlying rules, with test-takers required to infer these rules and apply them to novel test inputs. The benchmark uses colored grids (up to 30x30) with 10 discrete colors/symbols, designed to measure human-like general fluid intelligence and skill-acquisition efficiency with minimal prior knowledge.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arc-agi:o3-2025-04-16", "reported_at": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arc-agi", "languages": [], "modality": "image", "name": "ARC-AGI", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.0, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.95, "raw_min": 0.418, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arc-agi:gpt-5.5", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500"}, "unit": null}, "slug": "llm-stats-arc-agi", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500", "unit": null}, {"aliases": ["ARC-AGI", "ARC-AGI-1", "ARC-AGI-2"], "categories": ["reasoning"], "collected_at": null, "description": "Compute per task is reported alongside score by the maintainers and should not be dropped.", "evidence_summary": {"document_count": 3, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:arc_agi", "languages": [], "modality": null, "name": "ARC-AGI", "openness": "unknown", "publisher": null, "released": "2019-11-05", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:arc_agi", "source_url": "https://arcprize.org/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "arc_agi", "source": "model_reports", "source_url": "https://arcprize.org/", "unit": null}, {"aliases": [], "categories": ["reasoning", "spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ARC-AGI-2 is an upgraded benchmark for measuring abstract reasoning and problem-solving abilities in AI systems through visual grid transformation tasks. It evaluates fluid intelligence via input-output grid pairs (1x1 to 30x30) using colored cells (0-9), requiring models to identify underlying transformation rules from demonstration examples and apply them to test cases. Designed to be easy for humans but challenging for AI, focusing on core cognitive abilities like spatial reasoning, pattern recognition, and compositional generalization.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arc-agi-v2:o3-2025-04-16", "reported_at": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arc-agi-v2", "languages": [], "modality": "multimodal", "name": "ARC-AGI v2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.0, "display_multiplier": 100, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 0.85, "raw_min": 0.049, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arc-agi-v2:gpt-5.5", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-arc-agi-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500", "unit": null}, {"aliases": ["ARC-AGI-2", "ARC-AGI 2"], "categories": ["reasoning"], "collected_at": null, "description": "Cost per task is part of the official result and is routinely dropped when the score is quoted on its own.", "evidence_summary": {"document_count": 2, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000arc_agi_2\u0000google_gemini_3_1_pro_model_card\u0000arc_agi_2\u0000Thinking (High), ARC Prize Verified\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "first_score_reported_at": "2026-02-19", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000arc_agi_2\u0000google_gemini_3_1_pro_model_card\u0000arc_agi_2\u0000Thinking (High), ARC Prize Verified\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:arc_agi_2", "languages": [], "modality": null, "name": "ARC-AGI-2", "openness": "unknown", "publisher": null, "released": "2025-03-24", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:arc_agi_2", "source_url": "https://arcprize.org/arc-agi/2/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.1, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 77.1, "raw_min": 77.1, "source_reference": {"obs_id": "curated\u0000arc_agi_2\u0000google_gemini_3_1_pro_model_card\u0000arc_agi_2\u0000Thinking (High), ARC Prize Verified\u0000Gemini 3.1 Pro", "observation_id": "curated\u0000arc_agi_2\u0000google_gemini_3_1_pro_model_card\u0000arc_agi_2\u0000Thinking (High), ARC Prize Verified\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "reported_date": "2026-02-19", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "unit": "percent"}, "slug": "arc_agi_2", "source": "model_reports", "source_url": "https://arcprize.org/arc-agi/2/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ARC-AGI-3 is the third-generation Abstraction and Reasoning Corpus benchmark, an interactive-reasoning evaluation designed to measure fluid, novel problem-solving ability that remains far from saturated for frontier models.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arc-agi-3:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arc-agi-3", "languages": [], "modality": "multimodal", "name": "ARC-AGI-3", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 30.2, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.302, "raw_min": 0.0018, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arc-agi-3:claude-opus-5", "reported_date": "2026-07-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500"}, "unit": null}, "slug": "llm-stats-arc-agi-3", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500", "unit": null}, {"aliases": ["ARC-AGI-3", "ARC-AGI 3"], "categories": ["reasoning"], "collected_at": null, "description": "Interactive multi-step format, so the agent harness is part of the measurement. Scores are low and spread wide, which makes small absolute gaps look larger than they are.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000arc_agi_3\u0000anthropic_claude_opus_5_system_card\u0000arc_agi_3\u0000high effort, verified score, semi-private evaluation set\u0000Claude Opus 5", "reported_at": "2026-07-24", "source_url": "https://www.anthropic.com/news/claude-opus-5"}, "first_score_reported_at": "2026-07-24", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000arc_agi_3\u0000anthropic_claude_opus_5_system_card\u0000arc_agi_3\u0000high effort, verified score, semi-private evaluation set\u0000Claude Opus 5", "reported_at": "2026-07-24", "source_url": "https://www.anthropic.com/news/claude-opus-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:arc_agi_3", "languages": [], "modality": null, "name": "ARC-AGI-3", "openness": "unknown", "publisher": null, "released": "2026-01-15", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:arc_agi_3", "source_url": "https://arcprize.org/arc-agi/3/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 30.16, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 30.16, "raw_min": 30.16, "source_reference": {"obs_id": "curated\u0000arc_agi_3\u0000anthropic_claude_opus_5_system_card\u0000arc_agi_3\u0000high effort, verified score, semi-private evaluation set\u0000Claude Opus 5", "observation_id": "curated\u0000arc_agi_3\u0000anthropic_claude_opus_5_system_card\u0000arc_agi_3\u0000high effort, verified score, semi-private evaluation set\u0000Claude Opus 5", "reported_at": "2026-07-24", "reported_date": "2026-07-24", "source_id": "anthropic_claude_opus_5_system_card", "source_url": "https://www.anthropic.com/news/claude-opus-5"}, "unit": "percent"}, "slug": "arc_agi_3", "source": "model_reports", "source_url": "https://arcprize.org/arc-agi/3/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The AI2 Reasoning Challenge (ARC) Challenge Set is a multiple-choice question-answering benchmark containing grade-school level science questions that require advanced reasoning capabilities. ARC-C specifically contains questions that were answered incorrectly by both retrieval-based and word co-occurrence algorithms, making it a particularly challenging subset designed to test commonsense reasoning abilities in AI systems.", "evidence_summary": {"document_count": 1, "model_count": 34, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arc-c:claude-3-opus-20240229", "reported_at": "2024-02-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "llm-stats:arc-c", "languages": [], "modality": "text", "name": "ARC-C", "openness": "unknown", "publisher": null, "released": "2018-03-14", "released_reference": {"basis": "paper_first_version", "note": "The ARC introduction releases both the Challenge and Easy partitions.", "source_key": "opencompass:502", "source_url": "https://arxiv.org/abs/1803.05457"}, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 34, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.2, "display_multiplier": 100, "model_count": 34, "model_count_basis": "source_model_id", "numeric_count": 34, "raw_max": 0.972, "raw_min": 0.406, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arc-c:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500"}, "unit": null}, "slug": "llm-stats-arc-c", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The AI2’s Reasoning Challenge (ARC) dataset is a multiple-choice question-answering dataset, containing questions from science exams from grade 3 to grade 9. The dataset is split in two partitions: Easy and Challenge, where the latter partition contains the more difficult questions that require reasoning. Most of the questions have 4 answer choices, with <1% of all the questions having either 3 or 5 answer choices. AI2的推理挑战（ARC）数据集是一个多项选择问题回答数据集，包含了从三年级到九年级的科学考试中提取的问题。该数据集分为两个部分：简单和挑战，其中后者包含了需要推理能力的更难的问题。大多数问题有4个答案选项，仅有不到1％的问题有3个或5个答案选项。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:502", "languages": [], "modality": null, "name": "ARC-c", "openness": "unknown", "publisher": null, "released": "2018-03-14", "released_reference": {"basis": "paper_first_version", "note": "The ARC introduction releases both the Challenge and Easy partitions.", "source_key": "opencompass:502", "source_url": "https://arxiv.org/abs/1803.05457"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-502-arc-c", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ARC-c", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ARC-E (AI2 Reasoning Challenge - Easy Set) is a subset of grade-school level, multiple-choice science questions that requires knowledge and reasoning capabilities. Part of the AI2 Reasoning Challenge dataset containing 5,197 questions that test scientific reasoning and factual knowledge. The Easy Set contains questions that are answerable by retrieval-based and word co-occurrence algorithms, making them more accessible than the Challenge Set.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arc-e:gemma-2-27b-it", "reported_at": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arc-e", "languages": [], "modality": "text", "name": "ARC-E", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.6, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.886, "raw_min": 0.607, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arc-e:gemma-2-27b-it", "reported_date": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500"}, "unit": null}, "slug": "llm-stats-arc-e", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The AI2’s Reasoning Challenge (ARC) dataset is a multiple-choice question-answering dataset, containing questions from science exams from grade 3 to grade 9. The dataset is split in two partitions: Easy and Challenge, where the latter partition contains the more difficult questions that require reasoning. Most of the questions have 4 answer choices, with <1% of all the questions having either 3 or 5 answer choices. AI2的推理挑战（ARC）数据集是一个多项选择问题回答数据集，包含了从三年级到九年级的科学考试中提取的问题。该数据集分为两个部分：简单和挑战，其中后者包含了需要推理能力的更难的问题。大多数问题有4个答案选项，仅有不到1％的问题有3个或5个答案选项。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:503", "languages": [], "modality": null, "name": "ARC-e", "openness": "unknown", "publisher": null, "released": "2018-03-14", "released_reference": {"basis": "paper_first_version", "note": "The ARC introduction releases both the Easy and Challenge partitions.", "source_key": "opencompass:503", "source_url": "https://arxiv.org/abs/1803.05457"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-503-arc-e", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ARC-e", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ARC-AGI-2 is the second-generation Abstraction and Reasoning Corpus benchmark measuring fluid, general reasoning and abstraction.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arcagi2:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arcagi2", "languages": [], "modality": "text", "name": "ArcAGI2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.625, "raw_min": 0.613, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arcagi2:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500"}, "unit": null}, "slug": "llm-stats-arcagi2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "creativity", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Arena-Hard-Auto is an automatic evaluation benchmark for instruction-tuned LLMs consisting of 500 challenging real-world prompts curated by BenchBuilder. It includes open-ended software engineering problems, mathematical questions, and creative writing tasks. The benchmark uses LLM-as-a-Judge methodology with GPT-4.1 and Gemini-2.5 as automatic judges to approximate human preference. Arena-Hard achieves 98.6% correlation with human preference rankings and provides 3x higher separation of model performances compared to MT-Bench, making it highly effective for distinguishing between models of similar quality.", "evidence_summary": {"document_count": 1, "model_count": 26, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arena-hard:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:arena-hard", "languages": [], "modality": "text", "name": "Arena Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 26, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.6, "display_multiplier": 100, "model_count": 26, "model_count_basis": "source_model_id", "numeric_count": 26, "raw_max": 0.956, "raw_min": 0.267, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arena-hard:qwen3-235b-a22b", "reported_date": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-arena-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500", "unit": null}, {"aliases": ["Arena-Hard", "ArenaHard"], "categories": ["human_preference"], "collected_at": null, "description": "LLM-judge dependent, with known style and length bias.", "evidence_summary": {"document_count": 4, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000arena_hard\u0000deepseek_v3_report\u0000arena_hard\u0000GPT-4-1106 judge\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000arena_hard\u0000deepseek_v3_report\u0000arena_hard\u0000GPT-4-1106 judge\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:arena_hard", "languages": [], "modality": null, "name": "Arena-Hard", "openness": "unknown", "publisher": null, "released": "2024-04-19", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:arena_hard", "source_url": "https://github.com/lmarena/arena-hard-auto"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.6, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 95.6, "raw_min": 85.5, "source_reference": {"obs_id": "curated\u0000arena_hard\u0000qwen3_technical_report\u0000arena_hard\u0000thinking mode\u0000Qwen3-235B-A22B (Thinking)", "observation_id": "curated\u0000arena_hard\u0000qwen3_technical_report\u0000arena_hard\u0000thinking mode\u0000Qwen3-235B-A22B (Thinking)", "reported_at": "2025-05-14", "reported_date": "2025-05-14", "source_id": "qwen3_technical_report", "source_url": "https://arxiv.org/abs/2505.09388"}, "unit": "percent"}, "slug": "arena_hard", "source": "model_reports", "source_url": "https://github.com/lmarena/arena-hard-auto", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "general", "creativity", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Arena-Hard-Auto v2 is a challenging benchmark consisting of 500 carefully curated prompts sourced from Chatbot Arena and WildChat-1M, designed to evaluate large language models on real-world user queries. The benchmark covers diverse domains including open-ended software engineering problems, mathematics, creative writing, and technical problem-solving. It uses LLM-as-a-Judge for automatic evaluation, achieving 98.6% correlation with human preference rankings while providing 3x higher separation of model performances compared to MT-Bench. The benchmark emphasizes prompt specificity, complexity, and domain knowledge to better distinguish between model capabilities.", "evidence_summary": {"document_count": 1, "model_count": 16, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arena-hard-v2:qwen3-235b-a22b-instruct-2507", "reported_at": "2025-07-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arena-hard-v2", "languages": [], "modality": "text", "name": "Arena-Hard v2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 16, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.2, "display_multiplier": 100, "model_count": 16, "model_count_basis": "source_model_id", "numeric_count": 16, "raw_max": 0.862, "raw_min": 0.368, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arena-hard-v2:mimo-v2-flash", "reported_date": "2025-12-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-arena-hard-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Arena-Hard-Auto is an automated benchmark for instruction-tuned LLMs, designed to efficiently approximate human preferences. Arena-Hard-Auto 是一个用于评估 LLM 的基准，自动甄选 500 条高难度开放式提示，从模型区分度、人类偏好一致性与提示质量三维度进行严苛评测。依托 BenchBuilder 管道、主题建模与 LLM 裁判，实现众包数据→筛选→评分的全自动闭环。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2075", "languages": [], "modality": null, "name": "Arena-Hard-Auto", "openness": "open", "publisher": "University of California, Berkeley", "released": "2024-04-19", "released_reference": {"basis": "release_announcement", "note": "The original LMSYS announcement introduces Arena-Hard-Auto v0.1 and its leaderboard, before the June paper.", "source_key": "opencompass:2075", "source_url": "https://www.lmsys.org/blog/2024-04-19-arena-hard/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2075-arena-hard-auto", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Arena-Hard-Auto", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "视觉问答", "Visual-Qa", "物理智能", "Embodied AI", "逻辑推理", "图像理解", "Image Understanding", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "As Multimodal Large Language Models (MLLMs) continue to evolve, their cognitive and reasoning capabilities have seen remarkable progress. However, challenges in visual fine-grained perception and commonsense causal inference persist. This paper introduces Argus Inspection, a multimodal benchmark wit As Multimodal Large Language Models (MLLMs) continue to evolve, their cognitive and reasoning capabilities have seen remarkable progress. However, challenges in visual fine-grained perception and commonsense causal inference persist. This paper introduces Argus Inspection, a multimodal benchmark wit", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:2370", "languages": [], "modality": "multimodal", "name": "ArgusInspection", "openness": "unknown", "publisher": "Shanghai Artificial Intelligence Laboratory", "released": "2025-10-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2370-argusinspection", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ArgusInspection", "unit": null}, {"aliases": [], "categories": ["spatial_reasoning", "3d", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ARKitScenes evaluates 3D scene understanding and spatial reasoning in AR/VR contexts.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arkitscenes:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arkitscenes", "languages": [], "modality": "multimodal", "name": "ARKitScenes", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.537, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.537, "raw_min": 0.537, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arkitscenes:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500"}, "unit": null}, "slug": "llm-stats-arkitscenes", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500", "unit": null}, {"aliases": [], "categories": ["frontend_development", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Artifacts Bench evaluates a model's ability to generate visual code artifacts, measuring the quality of generated interactive and visual front-end outputs from natural-language requests.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:artifacts-bench:mai-code-1-flash", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:artifacts-bench", "languages": [], "modality": "text", "name": "Artifacts Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.51, "raw_min": 0.364, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:artifacts-bench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-artifacts-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "代码", "Code", "ArtifactsBench", "代码可视化", "代码生成", "多模态模型", "VLM", "逻辑推理", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ArtifactsBench is a benchmark with 1,825 tasks for evaluating LLM-generated visual and interactive code. It addresses the gap of traditional benchmarks that focus only on algorithmic correctness by assessing visual fidelity and user interaction. rtifactsBench 是一个专注于弥合传统代码评测中“视觉-交互”鸿沟的新型基准。它旨在全面评估大语言模型（LLM）生成动态可视化与交互式代码的能力，而非仅仅考核算法正确性。\n该基准包含1825个真实世界的任务，并开创了一套自动化多模态评估流程：系统会自动渲染代码、捕捉其视觉与交互行为，再由一个多模态大模型（MLLM）依据详细清单进行评分。该流程与人类专家判断的一致性高达94.4%，证明了其高度可靠性。\nArtifactsBench 已将数据集、评估框架及工具链完全开源，为社区提供了一个可扩展且精准的工具，以推动能创造丰富用户体验的新一代生成模型的发展。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2052", "languages": [], "modality": "multimodal", "name": "ArtifactsBench", "openness": "unknown", "publisher": "Tencent", "released": "2025-07-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2052-artifactsbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ArtifactsBench", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Artificial Analysis benchmark evaluates AI models across quality, speed, and pricing dimensions, providing a composite assessment of model capabilities for real-world usage.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:artificial-analysis:minimax-m2.7", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:artificial-analysis", "languages": [], "modality": "text", "name": "Artificial Analysis", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.0, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.59, "raw_min": 0.4, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:artificial-analysis:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500"}, "unit": null}, "slug": "llm-stats-artificial-analysis", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ArXivMath is a final-answer benchmark of research-level mathematics maintained by MathArena. Problems are extracted monthly from recent arXiv paper abstracts, then filtered through automated and manual checks to ensure they are self-contained, non-trivial, and verifiable. Because problems are drawn from active research, the benchmark is more realistic and more closely connected to mathematical research than contest or olympiad benchmarks.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:arxivmath:claude-sonnet-5", "reported_at": "2026-06-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:arxivmath", "languages": [], "modality": "text", "name": "ArXivMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.722, "raw_min": 0.522, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:arxivmath:claude-sonnet-5", "reported_date": "2026-06-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500"}, "unit": null}, "slug": "llm-stats-arxivmath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500", "unit": null}, {"aliases": [], "categories": ["math"], "collected_at": null, "description": "Competition-style mathematics drawn from arXiv; test-time compute budget is part of the result.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000arxivmath\u0000tencent_hy4_preview\u0000arxivmath\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000arxivmath\u0000tencent_hy4_preview\u0000arxivmath\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:arxivmath", "languages": [], "modality": null, "name": "ArXivMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.6, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 66.6, "raw_min": 66.6, "source_reference": {"obs_id": "curated\u0000arxivmath\u0000tencent_hy4_preview\u0000arxivmath\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000arxivmath\u0000tencent_hy4_preview\u0000arxivmath\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "arxivmath", "source": "model_reports", "source_url": "https://matharena.ai/arxivmath", "unit": "percent"}, {"aliases": [], "categories": ["数学", "Math", "大语言模型", "LLM", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ASDiv is a new MWP corpus that contains diverse lexicon patterns with wide problem type coverage. Each problem provides consistent equations and answers. It is further annotated with the corresponding problem type and grade level. ASDiv 是一个新的数学文字问题（MWP）语料库，包含多样的词汇模式，覆盖广泛的问题类型。每个问题提供对应的方程和答案。它进一步标注了相应的问题类型和年级水平，可用于测试系统的能力，并指明问题的难度等级。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1114", "languages": [], "modality": null, "name": "ASDiv", "openness": "restricted", "publisher": "Institute of Information Science, Academia Sinica", "released": "2020-07-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1114-asdiv", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ASDiv", "unit": null}, {"aliases": ["ASI-Bench", "ASI Bench"], "categories": ["science"], "collected_at": null, "description": "First benchmark that jointly evaluates general intelligence, innovation, and autonomous execution, via 60 project-level scientific research tasks spanning 11 domains, progressively withdrawing human guidance. Very new (arXiv 2608.17271, code at github.com/apexin-ai/ASI-Bench) and intended to probe superintelligence-level behaviour, so high variance across seeds and scaffolds should be expected before any vendor adopts it. Only the seed-31415 instances are public; the seed-42 references stay private for official leaderboard scoring, so a locally computed score and a leaderboard score are not the same measurement. This registry's maintainer is one of the benchmark's authors.", "evidence_summary": {"document_count": null, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:asi_bench", "languages": [], "modality": null, "name": "ASI-Bench", "openness": "unknown", "publisher": null, "released": "2026-08-18", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:asi_bench", "source_url": "https://arxiv.org/abs/2608.17271"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "asi_bench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2608.17271", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AssetOpsBench is a comprehensive benchmark designed to evaluate the performance of large language models (LLMs) and AI agents in complex asset operation and maintenance tasks. AssetOpsBench 是一个专注于评估大语言模型（LLM）和智能体在资产运维领域复杂任务中实际表现的多维度评测基准。该基准旨在检验模型在工业场景下的任务规划、多步推理、工具调用、安全合规性以及领域知识理解等核心能力，覆盖设备维护、异常诊断、风险评估等典型运维场景。测试集包含 1,000 个高质量样本，涉及 5 大类任务和 20 余种细分领域，数据来源于真实运维手册、工单记录及专家验证案例。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1991", "languages": [], "modality": null, "name": "AssetOpsBench", "openness": "unknown", "publisher": "IBMResearch-Yorktown , IBMResearch-Ireland", "released": "2025-06-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1991-assetopsbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AssetOpsBench", "unit": null}, {"aliases": [], "categories": [], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:community:ed90e889-4678-4fbd-98ab-0e654f4bf35e", "languages": [], "modality": null, "name": "atlas", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-community-ed90e889-4678-4fbd-98ab-0e654f4bf35e", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Aed90e889-4678-4fbd-98ab-0e654f4bf35e?top_n=500", "unit": null}, {"aliases": [], "categories": ["safety"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AttaQ is a unique dataset containing adversarial examples in the form of questions designed to provoke harmful or inappropriate responses from large language models. The benchmark evaluates safety vulnerabilities by using specialized clustering techniques that analyze both the semantic similarity of input attacks and the harmfulness of model responses, facilitating targeted improvements to model safety mechanisms.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:attaq:granite-3.3-8b-base", "reported_at": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:attaq", "languages": [], "modality": "text", "name": "AttaQ", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.5, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.885, "raw_min": 0.861, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:attaq:granite-3.3-8b-base", "reported_date": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500"}, "unit": null}, "slug": "llm-stats-attaq", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "语言", "Language", "安全", "Safety", "audio", "jailbreak", "LAM", "多模态模型", "VLM", "语言理解", "Comprehension", "安全对齐", "Safety Alignment", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LAMs face jailbreak risks. AJailBench, our new benchmark, reveals leading LAMs lack robustness. Subtle audio perturbations significantly degrade their safety. We release AJailBench for research. LAM 面临越狱风险。我们新的基准测试 AJailBench 揭示，领先的 LAM 缺乏稳健性。细微的音频干扰会显著降低其安全性。我们发布 AJailBench 进行研究。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1835", "languages": [], "modality": "multimodal", "name": "AudioJailbreak", "openness": "open", "publisher": "mbzuai", "released": "2025-05-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1835-audiojailbreak", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AudioJailbreak", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "安全", "Safety", "多模态模型", "VLM", "安全对齐", "Safety Alignment", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AudioTrust is a comprehensive trust evaluation framework for Audio Large Language Models (ALLMs) that effectively reveals potential risks in six dimensions: fairness, hallucination, security, privacy, robustness, and authentication. It aggregates over 4,420 real-world audio/text data samples, coveri AudioTrust针对Audio Large Language Models（ALLMs）的全方位可信评估框架，有效揭示音频大模型在公平性、幻觉、安全、隐私、鲁棒性和身份验证六大维度的潜在风险。汇集4,420+条真实场景音频/文本数据，覆盖日常对话、紧急呼叫、语音助手等18种实验设置，设计9项音频特定评测指标，构建自动化评估流水线。主要发现：闭源模型在鲁棒性和安全防护上表现更佳，开源模型对隐私和公平性仍存盲区；多数ALLMs对性别、口音、年龄等敏感属性存在系统性偏见。期待研究者基于AudioTrust继续优化音频大模型，共同推动更安全、可信的AI音频生态发展！", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1908", "languages": ["English"], "modality": "multimodal", "name": "AudioTrust", "openness": "open", "publisher": "Tsinghua University, Nanyang Technological University", "released": "2025-06-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1908-audiotrust", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AudioTrust", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "智能体", "Agent", "任务执行", "Task Execution", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AutoAdvExBench is a benchmark designed to evaluate large language models' (LLMs) ability to autonomously exploit adversarial example defenses, directly measuring LLMs' success on tasks regularly performed by machine learning security experts. AutoAdvExBench 是一个评估大型语言模型（LLMs）自主利用对抗性样本防御能力的基准，直接衡量LLMs在机器学习安全专家任务上的成功率。它主要评估模型理解学术论文、代码实现及生成对抗性攻击的能力。测试集包含75个对抗性样本防御实现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2078", "languages": [], "modality": null, "name": "AutoAdvExBench", "openness": "unknown", "publisher": "GoogleDeepMind , ETHZurich", "released": "2025-03-03", "released_reference": {"basis": "paper_first_version", "note": "First version introducing AutoAdvExBench and evaluating LLM agents on its defenses.", "source_key": "opencompass:2078", "source_url": "https://arxiv.org/abs/2503.01811"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2078-autoadvexbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AutoAdvExBench", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AutoLogi is an automated method for synthesizing open-ended logic puzzles to evaluate reasoning abilities of Large Language Models. The benchmark addresses limitations of existing multiple-choice reasoning evaluations by featuring program-based verification and controllable difficulty levels. It includes 1,575 English and 883 Chinese puzzles, enabling more reliable evaluation that better distinguishes models' reasoning capabilities across languages.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:autologi:kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:autologi", "languages": [], "modality": "text", "name": "AutoLogi", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.895, "raw_min": 0.895, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:autologi:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500"}, "unit": null}, "slug": "llm-stats-autologi", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AutomationBench is a tool-use benchmark that evaluates AI agents on automating real-world workflows, testing their ability to orchestrate tools and complete multi-step automation tasks.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:automationbench:claude-fable-5", "reported_at": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:automationbench", "languages": [], "modality": "text", "name": "AutomationBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.199999999999996, "display_multiplier": 100, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 0.482, "raw_min": 0.135, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:automationbench:glm-5.3", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500"}, "unit": null}, "slug": "llm-stats-automationbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500", "unit": null}, {"aliases": ["AutomationBench", "Zapier AutomationBench", "HLEAutomationBench"], "categories": ["agent"], "collected_at": null, "description": "Published by Zapier over its own automation surface, and cards report different task subsets of it.", "evidence_summary": {"document_count": 6, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000automationbench\u0000moonshot_kimi_k3_model_card\u0000automationbench\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "first_score_reported_at": "2026-06-13", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000automationbench\u0000moonshot_kimi_k3_model_card\u0000automationbench\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:automationbench", "languages": [], "modality": null, "name": "AutomationBench", "openness": "unknown", "publisher": null, "released": "2026-03-10", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:automationbench", "source_url": "https://github.com/zapier/automation-bench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.8, "display_multiplier": 1, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 48.8, "raw_min": 26.0, "source_reference": {"obs_id": "curated\u0000automationbench\u0000zai_glm_5_3_flash_model_card\u0000automationbench_v1.0.6\u0000AutomationBench v1.0.6, with null-type handling fix (PR\u0000GLM-5.3-Flash", "observation_id": "curated\u0000automationbench\u0000zai_glm_5_3_flash_model_card\u0000automationbench_v1.0.6\u0000AutomationBench v1.0.6, with null-type handling fix (PR\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "automationbench", "source": "model_reports", "source_url": "https://github.com/zapier/automation-bench", "unit": "percent"}, {"aliases": [], "categories": ["agentic", "tool-use", "business"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic SaaS workflows", "evidence_summary": {"document_count": 1, "model_count": 39, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:automationbench-aa:a6340098-d7ae-462d-b372-0a0a67fc44b4", "reported_at": "2025-10-15", "source_url": "https://artificialanalysis.ai/evaluations/automationbench-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:automationbench-aa", "languages": [], "modality": null, "name": "AutomationBench-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 39, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.74556199710793, "display_multiplier": 100, "model_count": 39, "model_count_basis": "source_model_id", "numeric_count": 39, "raw_max": 0.6274556199710792, "raw_min": 0.013565301670673362, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:automationbench-aa:b2331108-72ed-415a-82d1-188633875bbc", "reported_date": "2026-08-13", "source_url": "https://artificialanalysis.ai/evaluations/automationbench-aa"}, "unit": null}, "slug": "artificial-analysis-automationbench-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/automationbench-aa", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "AutomationBench-AA is Artificial Analysis's independently run version of AutomationBench, covering 657 real-world SaaS workflow tasks across 40 simulated applications (e.g. Gmail, Slack, Salesforce, HubSpot). It scores the share of objectives an agent completes without violating business guardrails, using a private held-out task set.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:automationbench-aa:grok-4.5", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:automationbench-aa", "languages": [], "modality": "text", "name": "AutomationBench-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.514, "raw_min": 0.514, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:automationbench-aa:grok-4.5", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500"}, "unit": null}, "slug": "llm-stats-automationbench-aa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "其他", "Other", "audio-visual", "多模态模型", "VLM", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AV-Odyssey Bench. This benchmark encompasses 26 different tasks and 4,555 carefully crafted problems, each incorporating text, visual, and audio components. All data are newly collected and annotated by humans, not from any existing audio-visual dataset. AV-Odyssey Bench. This benchmark encompasses 26 different tasks and 4,555 carefully crafted problems, each incorporating text, visual, and audio components. All data are newly collected and annotated by humans, not from any existing audio-visual dataset.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1396", "languages": [], "modality": "multimodal", "name": "AV-Odyssey-Bench", "openness": "open", "publisher": "CUHK MMLab, CUHK (SZ), Stanford University, UC Berkeley, Yale University", "released": "2024-12-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1396-av-odyssey-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AV-Odyssey-Bench", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AX-b is a broad-coverage diagnostic task, which requires to determine the logical relation between the given sentence pair, with three relations: entailment, contradiction and neutral. This task is selected from a subset of the GLUE broad-coverage diagnostic dataset, mainly used to test the model's understanding ability in grammar, semantics, world knowledge and so on. AX-b是一个广覆盖诊断任务，要求根据给定的句子对，判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。这个任务是从GLUE的广覆盖诊断数据集中选取了一部分数据，主要用来测试模型在语法、语义、世界知识等方面的理解能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:526", "languages": [], "modality": null, "name": "AX-b", "openness": "unknown", "publisher": null, "released": "2019-05-02", "released_reference": {"basis": "paper_first_version", "note": "The SuperGLUE first version introduces its collapsed broad-coverage diagnostic in section 3.2, derived from GLUE's original three-way diagnostic.", "source_key": "opencompass:526", "source_url": "https://arxiv.org/abs/1905.00537v1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-526-ax-b", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AX-b", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AX-g is a Winogender diagnostic task, which requires to determine which noun the pronoun refers to according to the given sentence and pronoun. This task is selected from a subset of the Winogender dataset, mainly used to test the model's ability in dealing with gender bias and discrimination. AX-g是一个Winogender诊断任务，要求根据给定的句子和代词，判断代词指代的是哪个名词。这个任务是从Winogender数据集中选取了一部分数据，主要用来测试模型在处理性别偏见和性别歧视方面的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:527", "languages": [], "modality": null, "name": "AX-g", "openness": "unknown", "publisher": null, "released": "2019-07-12", "released_reference": {"basis": "paper_publication", "note": "The first paper version containing SuperGLUE's filtered Winogender diagnostic is v2, dated July 12. V1 does not include this task.", "source_key": "opencompass:527", "source_url": "https://arxiv.org/abs/1905.00537v2"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-527-ax-g", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AX-g", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AXBENCH is a benchmark for large-scale evaluation of Language Model (LLM) control methods using synthetic data, focusing on fine-grained steering for safety and reliability. AXBENCH 是一个旨在评估LLM控制能力的基准。它通过概念检测和模型操控（包含概念、指令、流畅度）评估，旨在实现安全可靠的细粒度操控。基准使用大规模合成数据集，并集成了多种基线方法。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2084", "languages": [], "modality": null, "name": "AXBENCH", "openness": "unknown", "publisher": "Department of Computer Science,Stanford University , Pr(AI) R Group.", "released": "2025-01-28", "released_reference": {"basis": "paper_first_version", "note": "AxBench's own introduction, not later steering papers referenced by its README.", "source_key": "opencompass:2084", "source_url": "https://arxiv.org/abs/2501.17148"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2084-axbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AXBENCH", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "NeurIPS 2024", "大语言模型", "LLM", "长上下文", "Long Context", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BABILong is designed to test language models' ability to reason across facts distributed in extremely long documents. It contains a diverse set of 20 reasoning tasks, including fact chaining, simple induction, deduction, counting, and handling lists/sets. BABILong旨在测试语言模型对分布在极长文档中的事实进行推理的能力，涵盖事实链接、简单归纳、推导、计数和处理列表/集合等20种各类推理任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1266", "languages": [], "modality": null, "name": "BABILong", "openness": "restricted", "publisher": "AIRI", "released": "2024-06-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1266-babilong", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BABILong", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark for early-stage visual reasoning and perception on child-like vision tasks.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:babyvision:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:babyvision", "languages": [], "modality": "multimodal", "name": "BabyVision", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.7, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.857, "raw_min": 0.384, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:babyvision:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500"}, "unit": null}, "slug": "llm-stats-babyvision", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500", "unit": null}, {"aliases": ["BabyVision"], "categories": ["multimodal"], "collected_at": null, "description": "Multimodal vision benchmark. Temperature=1.0, top_p=0.95, max context 164K tokens. Images resized to shorter side at least 1.5K pixels.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000babyvision\u0000zai_glm_5_3_flash_model_card\u0000babyvision\u0000temp 1.0, top_p 0.95, max context 164K, images resized so shorter side >= 1.5K pixels\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000babyvision\u0000zai_glm_5_3_flash_model_card\u0000babyvision\u0000temp 1.0, top_p 0.95, max context 164K, images resized so shorter side >= 1.5K pixels\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:babyvision", "languages": [], "modality": null, "name": "BabyVision", "openness": "unknown", "publisher": null, "released": "2025-12-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:babyvision", "source_url": "https://github.com/babyvision/babyvision"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 53.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 53.4, "raw_min": 53.4, "source_reference": {"obs_id": "curated\u0000babyvision\u0000zai_glm_5_3_flash_model_card\u0000babyvision\u0000temp 1.0, top_p 0.95, max context 164K, images resized so shorter side >= 1.5K pixels\u0000GLM-5.3-Flash", "observation_id": "curated\u0000babyvision\u0000zai_glm_5_3_flash_model_card\u0000babyvision\u0000temp 1.0, top_p 0.95, max context 164K, images resized so shorter side >= 1.5K pixels\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "babyvision", "source": "model_reports", "source_url": "https://github.com/babyvision/babyvision", "unit": "percent"}, {"aliases": [], "categories": ["finance", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BankerToolBench is a public benchmark that evaluates models on banking and finance tool-use tasks. Models are scored against dataset rubrics, measuring their ability to correctly invoke tools and complete multi-step financial workflows.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bankertoolbench:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bankertoolbench", "languages": [], "modality": "text", "name": "BankerToolBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.12, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.7612, "raw_min": 0.7612, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bankertoolbench:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500"}, "unit": null}, "slug": "llm-stats-bankertoolbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500", "unit": null}, {"aliases": ["BankerToolBench", "Banker Tool Bench", "BTB"], "categories": ["professional"], "collected_at": null, "description": "End-to-end investment-banking tasks produce spreadsheets, presentations, and documents; the agent harness, financial-data tools, and rubric-grader configuration are part of the score.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000bankertoolbench\u0000tencent_hy4_preview\u0000bankertoolbench\u0000OpenCode scaffold, Gemini 3 Flash judge\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000bankertoolbench\u0000tencent_hy4_preview\u0000bankertoolbench\u0000OpenCode scaffold, Gemini 3 Flash judge\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:bankertoolbench", "languages": [], "modality": null, "name": "BankerToolBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.6, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 78.6, "raw_min": 78.6, "source_reference": {"obs_id": "curated\u0000bankertoolbench\u0000tencent_hy4_preview\u0000bankertoolbench\u0000OpenCode scaffold, Gemini 3 Flash judge\u0000Hy4 preview", "observation_id": "curated\u0000bankertoolbench\u0000tencent_hy4_preview\u0000bankertoolbench\u0000OpenCode scaffold, Gemini 3 Flash judge\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "bankertoolbench", "source": "model_reports", "source_url": "https://github.com/Handshake-AI-Research/bankertoolbench", "unit": "percent"}, {"aliases": [], "categories": ["math", "reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Big-Bench Hard (BBH) is a suite of 23 challenging tasks selected from BIG-Bench for which prior language model evaluations did not outperform the average human-rater. These tasks require multi-step reasoning across diverse domains including arithmetic, logical reasoning, reading comprehension, and commonsense reasoning. The benchmark was designed to test capabilities believed to be beyond current language models and focuses on evaluating complex reasoning skills including temporal understanding, spatial reasoning, causal understanding, and deductive logical reasoning.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bbh:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bbh", "languages": [], "modality": "text", "name": "BBH", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.87, "display_multiplier": 100, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 0.8887, "raw_min": 0.304, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bbh:qwen3-235b-a22b", "reported_date": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500"}, "unit": null}, "slug": "llm-stats-bbh", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BIG-Bench Hard (BBH) is a subset of the BIG-Bench, a diverse evaluation suite for language models. BBH focuses on a suite of 23 challenging tasks from BIG-Bench that were found to be beyond the capabilities of current language models. BIG Bench-Hard（BBH）是BIG Bench的一个子集，它是一个用于语言模型的多样化评估套件。BBH专注于BIG Bench的23项具有挑战性的任务，这些任务被发现超出了当前语言模型的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:539", "languages": [], "modality": null, "name": "BBH", "openness": "unknown", "publisher": null, "released": "2022-10-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-539-bbh", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BBH", "unit": null}, {"aliases": [], "categories": ["multimodal", "knowledge", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BC-VL is a vision-language benchmark for knowledge-grounded multimodal question answering.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bc-vl:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bc-vl", "languages": [], "modality": "multimodal", "name": "BC-VL", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.511, "raw_min": 0.511, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bc-vl:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500"}, "unit": null}, "slug": "llm-stats-bc-vl", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Beam 128K evaluates reasoning over long inputs at a 128K-token context length.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:beam-128k:muse-glimmer-30b", "reported_at": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:beam-128k", "languages": [], "modality": "text", "name": "Beam 128K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.10000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.651, "raw_min": 0.651, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:beam-128k:muse-glimmer-30b", "reported_date": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500"}, "unit": null}, "slug": "llm-stats-beam-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BELEBELE, a multiple-choice machine reading comprehension (MRC) dataset spanning 122 language variants. Significantly expanding the language coverage of natural language understanding (NLU) benchmarks, this dataset enables the evaluation of text models in\nhigh-, medium-, and low-resource languages. BELEBELE 是一个多项选择机器阅读理解（MRC）数据集，涵盖 122 种语言变体。该数据集显著扩展了自然语言理解（NLU）基准的语言覆盖范围，使得可以在高、中、低资源语言中评估文本模型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1086", "languages": [], "modality": null, "name": "Belebele", "openness": "unknown", "publisher": "FaceBook", "released": "2024-07-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1086-belebele", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Belebele", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BenchCAD is a benchmark for programmatic CAD reasoning built from 17,900 execution-verified CadQuery programs spanning 106 industrial part families, roughly half anchored to real ISO, DIN, EN, ASME, and IEC specification tables. It decomposes CAD capability into matched tasks; the Vision2Code task requires models to generate CadQuery code from multi-view renders, scored by voxel IoU against the reference geometry.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:benchcad:claude-sonnet-5", "reported_at": "2026-06-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:benchcad", "languages": [], "modality": "multimodal", "name": "BenchCAD", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.6, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.706, "raw_min": 0.373, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:benchcad:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500"}, "unit": null}, "slug": "llm-stats-benchcad", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BenchCAD variant evaluated with access to a Python tool for programmatic CAD reasoning.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:benchcad-with-python-tool:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:benchcad-with-python-tool", "languages": [], "modality": "multimodal", "name": "BenchCAD (with Python tool)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.39999999999999, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.834, "raw_min": 0.739, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:benchcad-with-python-tool:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500"}, "unit": null}, "slug": "llm-stats-benchcad-with-python-tool", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BenchMAX is a comprehensive, high-quality, and multiway parallel multilingual benchmark comprising 10 tasks designed to assess crucial capabilities across 17 diverse language. BenchMAX 是一个全面、高质量的多向并行多语言基准，包含 10 个任务，旨在评估 17 种不同语言的关键能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1562", "languages": ["English", "Chinese", "Japanese", "Korean", "French", "German", "Spanish", "Arabic", "Russian", "Vietnamese", "Thai", "Multilingual"], "modality": null, "name": "BenchMAX", "openness": "unknown", "publisher": "National Key Laboratory for Novel Software Technology, Nanjing University, etc.", "released": "2025-02-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1562-benchmax", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BenchMAX", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Beyond AIME is a difficult mathematical reasoning benchmark designed to test deeper reasoning chains and harder decomposition than standard AIME-style problem sets.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:beyond-aime:sarvam-105b", "reported_at": "2026-03-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:beyond-aime", "languages": [], "modality": "text", "name": "Beyond AIME", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.88, "raw_min": 0.583, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:beyond-aime:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500"}, "unit": null}, "slug": "llm-stats-beyond-aime", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The Berkeley Function Calling Leaderboard (BFCL) is the first comprehensive and executable function call evaluation dedicated to assessing Large Language Models' ability to invoke functions. It evaluates serial and parallel function calls across multiple programming languages (Python, Java, JavaScript, REST API) using a novel Abstract Syntax Tree (AST) evaluation method. The benchmark consists of over 2,000 question-function-answer pairs covering diverse application domains and complex use cases including multiple function calls, parallel function calls, and multi-turn interactions.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bfcl:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bfcl", "languages": [], "modality": "text", "name": "BFCL", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.5, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.885, "raw_min": 0.562, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bfcl:llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500"}, "unit": null}, "slug": "llm-stats-bfcl", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500", "unit": null}, {"aliases": ["BFCL", "Berkeley Function Calling Leaderboard"], "categories": ["tool_use"], "collected_at": null, "description": "Schema complexity and execution checking vary by version.", "evidence_summary": {"document_count": 4, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000bfcl\u0000qwen3_technical_report\u0000bfcl_v3\u0000v3\u0000Qwen3-235B-A22B (Thinking)", "reported_at": "2025-05-14", "source_url": "https://arxiv.org/abs/2505.09388"}, "first_score_reported_at": "2025-05-14", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000bfcl\u0000qwen3_technical_report\u0000bfcl_v3\u0000v3\u0000Qwen3-235B-A22B (Thinking)", "reported_at": "2025-05-14", "source_url": "https://arxiv.org/abs/2505.09388"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:bfcl", "languages": [], "modality": null, "name": "BFCL", "openness": "unknown", "publisher": null, "released": "2024-02-26", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:bfcl", "source_url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.9, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 72.9, "raw_min": 70.8, "source_reference": {"obs_id": "curated\u0000bfcl\u0000qwen3_5_model_card\u0000bfcl_v4\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000bfcl\u0000qwen3_5_model_card\u0000bfcl_v4\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "bfcl", "source": "model_reports", "source_url": "https://gorilla.cs.berkeley.edu/leaderboard.html", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "general", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Berkeley Function Calling Leaderboard (BFCL) v2 is a comprehensive benchmark for evaluating large language models' function calling capabilities. It features 2,251 question-function-answer pairs with enterprise and OSS-contributed functions, addressing data contamination and bias through live, user-contributed scenarios. The benchmark evaluates AST accuracy, executable accuracy, irrelevance detection, and relevance detection across multiple programming languages (Python, Java, JavaScript) and includes complex real-world function calling scenarios with multi-lingual prompts.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bfcl-v2:llama-3.2-3b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bfcl-v2", "languages": [], "modality": "text", "name": "BFCL v2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.3, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.773, "raw_min": 0.636, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bfcl-v2:llama-3.3-70b-instruct", "reported_date": "2024-12-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-bfcl-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "structured_output", "finance", "general", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Berkeley Function Calling Leaderboard v3 (BFCL-v3) is an advanced benchmark that evaluates large language models' function calling capabilities through multi-turn and multi-step interactions. It introduces extended conversational exchanges where models must retain contextual information across turns and execute multiple internal function calls for complex user requests. The benchmark includes 1000 test cases across domains like vehicle control, trading bots, travel booking, and file system management, using state-based evaluation to verify both system state changes and execution path correctness.", "evidence_summary": {"document_count": 1, "model_count": 19, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bfcl-v3:qwen3-coder-480b-a35b-instruct", "reported_at": "2025-01-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bfcl-v3", "languages": [], "modality": "text", "name": "BFCL-v3", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 19, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.8, "display_multiplier": 100, "model_count": 19, "model_count_basis": "source_model_id", "numeric_count": 19, "raw_max": 0.778, "raw_min": 0.63, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bfcl-v3:glm-4.5", "reported_date": "2025-07-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500"}, "unit": null}, "slug": "llm-stats-bfcl-v3", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Berkeley Function Calling Leaderboard V4 (BFCL-V4) evaluates LLMs on their ability to accurately call functions and APIs, including simple, multiple, parallel, and nested function calls across diverse programming scenarios.", "evidence_summary": {"document_count": 1, "model_count": 15, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bfcl-v4:nova-2-lite", "reported_at": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bfcl-v4", "languages": [], "modality": "text", "name": "BFCL-V4", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 15, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.0, "display_multiplier": 100, "model_count": 15, "model_count_basis": "source_model_id", "numeric_count": 15, "raw_max": 0.75, "raw_min": 0.253, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bfcl-v4:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500"}, "unit": null}, "slug": "llm-stats-bfcl-v4", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Berkeley Function Calling Leaderboard (BFCL) V3 MultiTurn benchmark that evaluates large language models' ability to handle multi-turn and multi-step function calling scenarios. The benchmark introduces complex interactions requiring models to manage sequential function calls, handle conversational context across multiple turns, and make dynamic decisions about when and how to use available functions. BFCL V3 uses state-based evaluation by verifying the actual state of API systems after function execution, providing more realistic assessment of function calling capabilities in agentic applications.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bfcl-v3-multiturn:nvidia-nemotron-nano-9b-v2", "reported_at": "2025-08-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bfcl-v3-multiturn", "languages": [], "modality": "text", "name": "BFCL_v3_MultiTurn", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.768, "raw_min": 0.669, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bfcl-v3-multiturn:minimax-m2.5", "reported_date": "2026-02-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500"}, "unit": null}, "slug": "llm-stats-bfcl-v3-multiturn", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Big Bench Audio is an audio reasoning benchmark adapted from a subset of Big Bench Hard, with text questions converted to spoken audio. It evaluates the reasoning ability of speech-to-speech and audio language models on tasks delivered as audio input, with accuracy scored by an independent evaluation (Artificial Analysis).", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:big-bench-audio:nova-2-sonic", "reported_at": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:big-bench-audio", "languages": [], "modality": "audio", "name": "Big Bench Audio", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.87, "raw_min": 0.87, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:big-bench-audio:nova-2-sonic", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500"}, "unit": null}, "slug": "llm-stats-big-bench-audio", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "finance", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Big Finance Bench evaluates models on complex financial-analysis tasks that require retrieving and reasoning over financial documents and performing multi-step quantitative work.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:big-finance-bench:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:big-finance-bench", "languages": [], "modality": "text", "name": "Big Finance Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 53.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.53, "raw_min": 0.36, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:big-finance-bench:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-big-finance-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Beyond the Imitation Game Benchmark (BIG-bench) is a collaborative benchmark consisting of 204+ tasks designed to probe large language models and extrapolate their future capabilities. It covers diverse domains including linguistics, mathematics, common-sense reasoning, biology, physics, social bias, software development, and more. The benchmark focuses on tasks believed to be beyond current language model capabilities and includes both English and non-English tasks across multiple languages.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:big-bench:gemini-1.0-pro", "reported_at": "2024-02-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:big-bench", "languages": [], "modality": "text", "name": "BIG-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.75, "raw_min": 0.682, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:big-bench:gemini-1.0-pro", "reported_date": "2024-02-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-big-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BIG-Bench Extra Hard (BBEH) is a challenging benchmark that replaces each task in BIG-Bench Hard with a novel task that probes similar reasoning capabilities but exhibits significantly increased difficulty. The benchmark contains 23 tasks testing diverse reasoning skills including many-hop reasoning, causal understanding, spatial reasoning, temporal arithmetic, geometric reasoning, linguistic reasoning, logic puzzles, and humor understanding. Designed to address saturation on existing benchmarks where state-of-the-art models achieve near-perfect scores, BBEH shows substantial room for improvement with best models achieving only 9.8-44.8% average accuracy.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:big-bench-extra-hard:gemma-3-12b-it", "reported_at": "2025-03-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:big-bench-extra-hard", "languages": [], "modality": "text", "name": "BIG-Bench Extra Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.4, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.744, "raw_min": 0.072, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:big-bench-extra-hard:gemma-4-31b-it", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-big-bench-extra-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500", "unit": null}, {"aliases": ["BigBench Extra Hard", "BBEH", "BIG-Bench Extra Hard"], "categories": ["reasoning"], "collected_at": null, "description": "Successor to BBH after saturation; per-task variance is high.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000bigbench_extra_hard\u0000google_gemma_4_model_card\u0000bigbench_extra_hard\u0000as reported\u0000Gemma 4 (31B)", "reported_at": "2026-03-11", "source_url": "https://huggingface.co/google/gemma-4-31B-it"}, "first_score_reported_at": "2026-03-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000bigbench_extra_hard\u0000google_gemma_4_model_card\u0000bigbench_extra_hard\u0000as reported\u0000Gemma 4 (31B)", "reported_at": "2026-03-11", "source_url": "https://huggingface.co/google/gemma-4-31B-it"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:bigbench_extra_hard", "languages": [], "modality": null, "name": "BIG-Bench Extra Hard", "openness": "unknown", "publisher": null, "released": "2025-02-26", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:bigbench_extra_hard", "source_url": "https://arxiv.org/abs/2502.19187"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 74.4, "raw_min": 74.4, "source_reference": {"obs_id": "curated\u0000bigbench_extra_hard\u0000google_gemma_4_model_card\u0000bigbench_extra_hard\u0000as reported\u0000Gemma 4 (31B)", "observation_id": "curated\u0000bigbench_extra_hard\u0000google_gemma_4_model_card\u0000bigbench_extra_hard\u0000as reported\u0000Gemma 4 (31B)", "reported_at": "2026-03-11", "reported_date": "2026-03-11", "source_id": "google_gemma_4_model_card", "source_url": "https://huggingface.co/google/gemma-4-31B-it"}, "unit": "percent"}, "slug": "bigbench_extra_hard", "source": "model_reports", "source_url": "https://arxiv.org/abs/2502.19187", "unit": "percent"}, {"aliases": [], "categories": ["math", "reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BIG-Bench Hard (BBH) is a subset of 23 challenging BIG-Bench tasks selected because prior language model evaluations did not outperform average human-rater performance. The benchmark contains 6,511 evaluation examples testing various forms of multi-step reasoning including arithmetic, logical reasoning (Boolean expressions, logical deduction), geometric reasoning, temporal reasoning, and language understanding. Tasks require capabilities such as causal judgment, object counting, navigation, pattern recognition, and complex problem solving.", "evidence_summary": {"document_count": 1, "model_count": 21, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:big-bench-hard:claude-3-opus-20240229", "reported_at": "2024-02-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:big-bench-hard", "languages": [], "modality": "text", "name": "BIG-Bench Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 21, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.10000000000001, "display_multiplier": 100, "model_count": 21, "model_count_basis": "source_model_id", "numeric_count": 21, "raw_max": 0.931, "raw_min": 0.391, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:big-bench-hard:claude-3-5-sonnet-20240620", "reported_date": "2024-06-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-big-bench-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark that challenges LLMs to invoke multiple function calls as tools from 139 libraries and 7 domains for 1,140 fine-grained programming tasks. Evaluates code generation with diverse function calls and complex instructions, featuring two variants: Complete (code completion based on comprehensive docstrings) and Instruct (generating code from natural language instructions).", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bigcodebench:qwen-2.5-coder-7b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bigcodebench", "languages": [], "modality": "text", "name": "BigCodeBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 45.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.454, "raw_min": 0.41, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bigcodebench:gemini-diffusion", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500"}, "unit": null}, "slug": "llm-stats-bigcodebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "代码", "Code", "大语言模型", "LLM", "逻辑推理", "Reasoning", "代码工程", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BigCodeBench is a benchmark that challenges LLMs to invoke multiple function calls as tools from 139 libraries and 7 domains for 1,140 fine-grained tasks. BigCodeBench用于评估LLM的代码生成能力，包含1140个可以调用139个库和7个域的多个函数来完成的细粒度任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1253", "languages": [], "modality": null, "name": "BigCodeBench", "openness": "open", "publisher": "Monash University", "released": "2024-06-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1253-bigcodebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BigCodeBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive benchmark that evaluates large language models' ability to solve complex, practical programming tasks via code generation. Contains 1,140 fine-grained tasks across 7 domains using function calls from 139 libraries. Challenges LLMs to invoke multiple function calls as tools and handle complex instructions for realistic software engineering and general-purpose reasoning tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bigcodebench-full:qwen-2.5-coder-32b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bigcodebench-full", "languages": [], "modality": "text", "name": "BigCodeBench-Full", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 49.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.496, "raw_min": 0.496, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bigcodebench-full:qwen-2.5-coder-32b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500"}, "unit": null}, "slug": "llm-stats-bigcodebench-full", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BigCodeBench-Hard is a subset of 148 challenging programming tasks from BigCodeBench, designed to evaluate large language models' ability to solve complex, real-world programming problems. These tasks require diverse function calls from multiple libraries across 7 domains including computation, networking, data analysis, and visualization. The benchmark tests compositional reasoning and the ability to implement complex instructions that span 139 libraries with an average of 2.8 libraries per task.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bigcodebench-hard:qwen-2.5-coder-32b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bigcodebench-hard", "languages": [], "modality": "text", "name": "BigCodeBench-Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 27.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.27, "raw_min": 0.27, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bigcodebench-hard:qwen-2.5-coder-32b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-bigcodebench-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BigO(Bench)是一个包含约 300 个需要用 Python 解决的代码问题的基准测试，以及 3,105 个编码问题和 1,190,250 个解决方案用于训练，以评估LLMs能否找到代码解决方案的时间-空间复杂度，或者生成符合时间-空间复杂度要求的代码解决方案。 BigO(Bench)是一个包含约 300 个需要用 Python 解决的代码问题的基准测试，以及 3,105 个编码问题和 1,190,250 个解决方案用于训练，以评估LLMs能否找到代码解决方案的时间-空间复杂度，或者生成符合时间-空间复杂度要求的代码解决方案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1680", "languages": [], "modality": null, "name": "BigOBench", "openness": "restricted", "publisher": "facebook", "released": "2025-03-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1680-bigobench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BigOBench", "unit": null}, {"aliases": [], "categories": ["safety", "healthcare", "biology"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BioLP-Bench is a model-graded evaluation measuring ability to find and correct mistakes in common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:biolp-bench:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:biolp-bench", "languages": [], "modality": "text", "name": "BioLP-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 37.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.37, "raw_min": 0.37, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:biolp-bench:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-biolp-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "biology"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BioMysteryBench evaluates a model's ability to reason through challenging molecular biology problems, reporting performance on a hard subset and on the subset of problems solved by human experts.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:biomysterybench:claude-fable-5", "reported_at": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:biomysterybench", "languages": [], "modality": "text", "name": "BioMysteryBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.10000000000001, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.901, "raw_min": 0.435, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:biomysterybench:claude-opus-5", "reported_date": "2026-07-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500"}, "unit": null}, "slug": "llm-stats-biomysterybench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500", "unit": null}, {"aliases": ["BioMysteryBench"], "categories": ["biology"], "collected_at": null, "description": "Reported in two splits (\"hard\" and \"human solved\") that differ by more than 35 points, so a bare score is unreadable without its split. Anthropic notes its own safety refusals depress this number, which means the score mixes capability with policy. First-party to Anthropic, which built it and publishes the dataset. The task set was revised after an answer-key audit, so the item count moved and scores are not comparable across dataset versions.", "evidence_summary": {"document_count": 2, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000biomysterybench\u0000tencent_hy4_preview\u0000biomysterybench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000biomysterybench\u0000tencent_hy4_preview\u0000biomysterybench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:biomysterybench", "languages": [], "modality": null, "name": "BioMysteryBench", "openness": "unknown", "publisher": null, "released": "2026-04-29", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:biomysterybench", "source_url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-full"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.3, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 71.3, "raw_min": 71.3, "source_reference": {"obs_id": "curated\u0000biomysterybench\u0000tencent_hy4_preview\u0000biomysterybench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000biomysterybench\u0000tencent_hy4_preview\u0000biomysterybench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "biomysterybench", "source": "model_reports", "source_url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-full", "unit": "percent"}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQLs) is a comprehensive text-to-SQL benchmark containing 12,751 question-SQL pairs across 95 databases (33.4 GB total) spanning 37+ professional domains. It evaluates large language models' ability to convert natural language to executable SQL queries in real-world scenarios with complex database schemas and dirty data.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bird-sql-(dev):gemini-2.0-flash", "reported_at": "2024-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bird-sql-(dev)", "languages": [], "modality": "text", "name": "Bird-SQL (dev)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.4, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.574, "raw_min": 0.064, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bird-sql-(dev):gemini-2.0-flash-lite", "reported_date": "2025-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500"}, "unit": null}, "slug": "llm-stats-bird-sql-dev", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BixBench is a benchmark for real-world bioinformatics and computational biology data analysis. It evaluates AI models on multi-step scientific workflows that require code execution, statistical reasoning, and biological domain knowledge to interpret experimental data.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:bixbench:gpt-5.5", "reported_at": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:bixbench", "languages": [], "modality": "text", "name": "BixBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.805, "raw_min": 0.805, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:bixbench:gpt-5.5", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500"}, "unit": null}, "slug": "llm-stats-bixbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "spatial_reasoning", "3d", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BLINK: Multimodal Large Language Models Can See but Not Perceive. A benchmark for multimodal language models focusing on core visual perception abilities. Reformats 14 classic computer vision tasks into 3,807 multiple-choice questions paired with single or multiple images and visual prompting. Tasks include relative depth estimation, visual correspondence, forensics detection, multi-view reasoning, counting, object localization, and spatial reasoning that humans can solve 'within a blink'.", "evidence_summary": {"document_count": 1, "model_count": 15, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:blink:phi-4-multimodal-instruct", "reported_at": "2025-02-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:blink", "languages": [], "modality": "multimodal", "name": "BLINK", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 15, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.39999999999999, "display_multiplier": 100, "model_count": 15, "model_count_basis": "source_model_id", "numeric_count": 15, "raw_max": 0.814, "raw_min": 0.527, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:blink:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500"}, "unit": null}, "slug": "llm-stats-blink", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BLINK focuses on MLLMs' core visual perception abilities. It contains 3,807 multiple-choice questions spanning 14 classic computer vision tasks. BLINK用于评估多模态大模型的视觉感知能力，包含来自14个经典计算机视觉任务的3807道多项选择题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1365", "languages": [], "modality": "multimodal", "name": "BLINK", "openness": "open", "publisher": "University of Pensylvania", "released": "2024-04-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1365-blink", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BLINK", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Blueprint-Bench 2 is an agentic spatial reasoning benchmark that evaluates a model's ability to understand, plan, and reason over architectural blueprints and other structured spatial documents. Scores are reported as a normalized score.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:blueprint-bench-2:gemini-3.5-flash", "reported_at": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:blueprint-bench-2", "languages": [], "modality": "multimodal", "name": "Blueprint-Bench 2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 38.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.386, "raw_min": 0.336, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:blueprint-bench-2:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500"}, "unit": null}, "slug": "llm-stats-blueprint-bench-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500", "unit": null}, {"aliases": ["Blueprint-Bench 2", "Blueprint-Bench", "BlueprintBench 2"], "categories": ["spatial_reasoning"], "collected_at": null, "description": "Spatial-reasoning set: 50 apartments, ~20 photos each, scored by a connectivity-graph grader, with a public leaderboard on the Andon Labs eval page (run in-house; dataset not openly downloadable). Builds on the original Blueprint-Bench paper (arXiv:2509.25229), a distinct instrument released 2025-09-24 whose scores must not be compared across versions.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:blueprint_bench_2", "languages": [], "modality": null, "name": "Blueprint-Bench 2", "openness": "unknown", "publisher": null, "released": "2026-05-04", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:blueprint_bench_2", "source_url": "https://andonlabs.com/evals/blueprint-bench-2"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "blueprint_bench_2", "source": "model_reports", "source_url": "https://andonlabs.com/evals/blueprint-bench-2", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BoolQ is a reading comprehension dataset for yes/no questions containing 15,942 naturally occurring examples. Each example consists of a question, passage, and boolean answer, where questions are generated in unprompted and unconstrained settings. The dataset challenges models with complex, non-factoid information requiring entailment-like inference to solve.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:boolq:gemma-2-27b-it", "reported_at": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:boolq", "languages": [], "modality": "text", "name": "BoolQ", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.03999999999999, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.8804, "raw_min": 0.764, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:boolq:hermes-3-70b", "reported_date": "2024-08-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500"}, "unit": null}, "slug": "llm-stats-boolq", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BoolQ is a question answering dataset for yes/no questions containing 15942 examples. These questions are naturally occurring ---they are generated in unprompted and unconstrained settings. Each example is a triplet of (question, passage, answer), with the title of the page as optional additional context. BoolQ是一个包含15942个示例的是/否问题的问答数据集。这些问题是自然生成的——即在无prompt和无约束的环境中产生的。每个例子都是一个三元组(问题、段落、答案)，页面标题是可选的附加上下文。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:510", "languages": [], "modality": null, "name": "BoolQ", "openness": "unknown", "publisher": null, "released": "2019-05-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-510-boolq", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BoolQ", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "大语言模型", "LLM", "检索能力", "Retrieval", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BRIGHT is the first text retrieval benchmark that requires intensive reasoning to retrieve relevant documents. BRIGHT 是第一个需要大量推理来检索相关文档的文本检索基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1571", "languages": [], "modality": null, "name": "BRIGHT", "openness": "unknown", "publisher": "The University of Hong Kong, etc.", "released": "2024-10-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1571-bright", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BRIGHT", "unit": null}, {"aliases": [], "categories": ["math"], "collected_at": null, "description": "arXiv proofs with planted errors; tests whether a model catches a broken argument instead of reproducing it.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000brokenarxiv\u0000tencent_hy4_preview\u0000brokenarxiv\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000brokenarxiv\u0000tencent_hy4_preview\u0000brokenarxiv\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:brokenarxiv", "languages": [], "modality": null, "name": "BrokenArXiv", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 54.6, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 54.6, "raw_min": 54.6, "source_reference": {"obs_id": "curated\u0000brokenarxiv\u0000tencent_hy4_preview\u0000brokenarxiv\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000brokenarxiv\u0000tencent_hy4_preview\u0000brokenarxiv\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "brokenarxiv", "source": "model_reports", "source_url": "https://matharena.ai/brokenarxiv", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "search", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BrowseComp is a benchmark comprising 1,266 questions that challenge AI agents to persistently navigate the internet in search of hard-to-find, entangled information. The benchmark measures agents' ability to exercise persistence in information gathering, demonstrate creativity in web navigation, and find concise, verifiable answers. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers.", "evidence_summary": {"document_count": 1, "model_count": 62, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:browsecomp:deepseek-v3.1", "reported_at": "2025-01-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:browsecomp", "languages": [], "modality": "text", "name": "BrowseComp", "openness": "restricted", "publisher": "OpenAI", "released": "2025-04-10", "released_reference": null, "repo_kind": "harness_only", "repo_resolution_status": "resolved", "score_count": 62, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.2, "display_multiplier": 100, "model_count": 62, "model_count_basis": "source_model_id", "numeric_count": 62, "raw_max": 0.912, "raw_min": 0.089, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:browsecomp:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500"}, "unit": null}, "slug": "llm-stats-browsecomp", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500", "unit": null}, {"aliases": ["BrowseComp"], "categories": ["agent"], "collected_at": null, "description": "Live web. The result depends on what the internet contained on the day of the run.", "evidence_summary": {"document_count": 12, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000browsecomp\u0000zai_glm_5_model_card\u0000browsecomp\u0000no context management, retain most recent 5 turns\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000browsecomp\u0000zai_glm_5_model_card\u0000browsecomp\u0000no context management, retain most recent 5 turns\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:browsecomp", "languages": [], "modality": null, "name": "BrowseComp", "openness": "unknown", "publisher": null, "released": "2025-04-10", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:browsecomp", "source_url": "https://openai.com/index/browsecomp/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.2, "display_multiplier": 1, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 91.2, "raw_min": 62.0, "source_reference": {"obs_id": "curated\u0000browsecomp\u0000moonshot_kimi_k3_model_card\u0000browsecomp\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000browsecomp\u0000moonshot_kimi_k3_model_card\u0000browsecomp\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "browsecomp", "source": "model_reports", "source_url": "https://openai.com/index/browsecomp/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "search"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A challenging benchmark for evaluating web browsing agents' ability to persistently navigate the internet and find hard-to-locate, entangled information. Comprises 1,266 questions requiring strategic reasoning, creative search, and interpretation of retrieved content, with short and easily verifiable answers.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:browsecomp-long-128k:gpt-5-2025-08-07", "reported_at": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:browsecomp-long-128k", "languages": [], "modality": "text", "name": "BrowseComp Long Context 128k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.0, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.92, "raw_min": 0.9, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:browsecomp-long-128k:gpt-5.2-2025-12-11", "reported_date": "2025-12-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500"}, "unit": null}, "slug": "llm-stats-browsecomp-long-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "search"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BrowseComp is a benchmark for measuring the ability of agents to browse the web, comprising 1,266 questions that require persistently navigating the internet in search of hard-to-find, entangled information. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers. The benchmark focuses on questions where answers are obscure, time-invariant, and well-supported by evidence scattered across the open web.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:browsecomp-long-256k:gpt-5-2025-08-07", "reported_at": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:browsecomp-long-256k", "languages": [], "modality": "text", "name": "BrowseComp Long Context 256k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.898, "raw_min": 0.888, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:browsecomp-long-256k:gpt-5.2-2025-12-11", "reported_date": "2025-12-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500"}, "unit": null}, "slug": "llm-stats-browsecomp-long-256k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "search", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "BrowseComp-VL is the vision-language variant of BrowseComp, evaluating multimodal models on web browsing comprehension tasks that require processing visual web page content alongside text.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:browsecomp-vl:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:browsecomp-vl", "languages": [], "modality": "multimodal", "name": "BrowseComp-VL", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.519, "raw_min": 0.519, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:browsecomp-vl:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500"}, "unit": null}, "slug": "llm-stats-browsecomp-vl", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "search"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A high-difficulty benchmark purpose-built to comprehensively evaluate LLM agents on the Chinese web, consisting of 289 multi-hop questions spanning 11 diverse domains including Film & TV, Technology, Medicine, and History. Questions are reverse-engineered from short, objective, and easily verifiable answers, requiring sophisticated reasoning and information reconciliation beyond basic retrieval. The benchmark addresses linguistic, infrastructural, and censorship-related complexities in Chinese web environments.", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:browsecomp-zh:deepseek-v3.1", "reported_at": "2025-01-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:browsecomp-zh", "languages": [], "modality": "text", "name": "BrowseComp-zh", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.3, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.703, "raw_min": 0.357, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:browsecomp-zh:qwen3.5-397b-a17b", "reported_date": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500"}, "unit": null}, "slug": "llm-stats-browsecomp-zh", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500", "unit": null}, {"aliases": ["BrowseComp-ZH", "BrowseComp-zh", "BrowseComp-Zh"], "categories": ["agent"], "collected_at": null, "description": "Chinese-language live web. Same day-to-day web drift as BrowseComp, over a different index.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000browsecomp_zh\u0000zai_glm_5_model_card\u0000browsecomp_zh\u0000no context management, retain most recent 5 turns\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000browsecomp_zh\u0000zai_glm_5_model_card\u0000browsecomp_zh\u0000no context management, retain most recent 5 turns\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:browsecomp_zh", "languages": [], "modality": null, "name": "BrowseComp-ZH", "openness": "unknown", "publisher": null, "released": "2025-04-27", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:browsecomp_zh", "source_url": "https://arxiv.org/abs/2504.19314"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.7, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 72.7, "raw_min": 70.3, "source_reference": {"obs_id": "curated\u0000browsecomp_zh\u0000zai_glm_5_model_card\u0000browsecomp_zh\u0000no context management, retain most recent 5 turns\u0000GLM-5", "observation_id": "curated\u0000browsecomp_zh\u0000zai_glm_5_model_card\u0000browsecomp_zh\u0000no context management, retain most recent 5 turns\u0000GLM-5", "reported_at": "2026-02-11", "reported_date": "2026-02-11", "source_id": "zai_glm_5_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "unit": "percent"}, "slug": "browsecomp_zh", "source": "model_reports", "source_url": "https://arxiv.org/abs/2504.19314", "unit": "percent"}, {"aliases": [], "categories": ["理解", "Understanding", "NAACL 2024", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "BUST is a comprehensive benchmark for evaluating synthetic text detectors, focusing on their effectiveness against outputs from various Large Language Models (LLMs). BUST 是一个综合基准，旨在评估合成文本检测器，BUST 使用多种指标来评估检测器，包括语言特征、可读性和作者态度。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1146", "languages": [], "modality": null, "name": "BUST", "openness": "unknown", "publisher": "Dalle Molle Institute for Artificial Intelligence Research (IDSIA)", "released": "2024-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1146-bust", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BUST", "unit": null}, {"aliases": [], "categories": ["创作", "Creation", "指令跟随", "Instruct", "多模态模型", "VLM", "视觉生成", "Visual Generation", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ByteMorph is a benchmark for instruction-guided image editing, focusing on evaluating models’ capabilities in handling non-rigid motions such as camera viewpoint changes, object deformations, human articulations, and complex interactions. ByteMorph 是一个面向指令驱动图像编辑的基准，专注于评估模型在处理非刚性运动（如相机视角变化、物体变形、人类动作和复杂交互）方面的能力。 该基准包括超过 600 万对高分辨率图像编辑样本，涵盖多种动态编辑场景，支持对模型在多种非刚性运动类型下的表现进行细粒度评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1983", "languages": [], "modality": null, "name": "ByteMorph", "openness": "open", "publisher": "ByteDance Seed , University of Southern California , University of Tokyo , etc.", "released": "2025-06-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1983-bytemorph", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ByteMorph", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "C-Eval is a comprehensive Chinese evaluation suite designed to assess advanced knowledge and reasoning abilities of foundation models in a Chinese context. It comprises 13,948 multiple-choice questions across 52 diverse disciplines spanning humanities, science, and engineering, with four difficulty levels: middle school, high school, college, and professional. The benchmark includes C-Eval Hard, a subset of very challenging subjects requiring advanced reasoning abilities.", "evidence_summary": {"document_count": 1, "model_count": 18, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:c-eval:qwen2-72b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:c-eval", "languages": [], "modality": "text", "name": "C-Eval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 18, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.30000000000001, "display_multiplier": 100, "model_count": 18, "model_count_basis": "source_model_id", "numeric_count": 18, "raw_max": 0.933, "raw_min": 0.407, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:c-eval:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500"}, "unit": null}, "slug": "llm-stats-c-eval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "C-Eval is a comprehensive Chinese evaluation suite for foundation models. It consists of 13948 multi-choice questions spanning 52 diverse disciplines and four difficulty levels. C-Eval 是一个全面的中文基础模型评估套件。它包含了13948个多项选择题，涵盖了52个不同的学科和四个难度级别。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:496", "languages": ["English", "Chinese"], "modality": null, "name": "C-Eval", "openness": "unknown", "publisher": null, "released": "2023-05-15", "released_reference": {"basis": "paper_first_version", "note": "First version of the paper introducing C-Eval.", "source_key": "opencompass:496", "source_url": "https://arxiv.org/abs/2305.08322"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-496-c-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/C-Eval", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "C-FAITH, a Chinese QA hallucination benchmark created from 1,399 knowledge documents obtained from web scraping, totaling 60,702 entries. C-FAITH，这是一个中国 QA 幻觉基准，由从网络抓取中获得的 1,399 份知识文档创建，总共 60,702 个条目。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1780", "languages": ["Chinese"], "modality": null, "name": "C-FAITH", "openness": "unknown", "publisher": "PKU", "released": "2025-04-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1780-c-faith", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/C-FAITH", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A free-form multiple-Choice Chinese machine reading Comprehension dataset (C3), containing 13,369 documents (dialogues or more formally written mixed-genre texts) and their associated 19,577 multiple-choice free-form questions collected from Chinese-as-a-second-language examinations 一个自由形式的多项选择中文机器阅读理解数据集（C3），包含13369篇文献（对话或更正式的混合体裁文本）及其相关的19577道自由选择题，这些问题都是从汉语作为第二语言的考试中收集到的", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:514", "languages": ["Chinese"], "modality": null, "name": "C3", "openness": "unknown", "publisher": null, "released": "2019-04-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-514-c3", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/C3", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CaLM is the first comprehensive benchmark for evaluating the causal reasoning capabilities of language models. The CaLM framework establishes a foundational taxonomy consisting of four modules: causal target, adaptation, metric, and error. CaLM是上海人工智能实验室联合同济大学、上海交通大学、北京大学及商汤科技发布首个大模型因果推理开放评测体系。首次从因果推理角度提出评估框架，为AI研究者打造可靠评测工具，从而为推进大模型认知能力向人类水平看齐提供指标参考。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1052", "languages": ["English", "Chinese"], "modality": null, "name": "CaLM", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-05-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1052-calm", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CaLM", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Capture-the-Flag Challenges is OpenAI's internal expansion of competitive, professional-level cybersecurity capture-the-flag tasks used to evaluate vulnerability identification and exploitation capability under the Preparedness Framework.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:capture-the-flag-challenges:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:capture-the-flag-challenges", "languages": [], "modality": "text", "name": "Capture-the-Flag Challenges (Internal)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.967, "raw_min": 0.852, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:capture-the-flag-challenges:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500"}, "unit": null}, "slug": "llm-stats-capture-the-flag-challenges", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "VQA", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CausalVQA tests causal reasoning in videos across five question types, and state-of-the-art multimodal models still trail human performance. CausalVQA 是面向视频问答的因果推理基准，涵盖反事实、假设、预判、规划、描述五类问题，强调真实物理场景。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1975", "languages": [], "modality": "multimodal", "name": "CausalVQA", "openness": "unknown", "publisher": "FAIR at Meta", "released": "2025-06-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1975-causalvqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CausalVQA", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Curriculum Learning of Bayesian Network Structures (CBNSL) benchmark for evaluating algorithms that learn Bayesian network structures from data using curriculum learning techniques. The benchmark uses networks from the bnlearn repository and evaluates structure learning performance using BDeu scoring metrics.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cbnsl:kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cbnsl", "languages": [], "modality": "text", "name": "CBNSL", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.956, "raw_min": 0.956, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cbnsl:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500"}, "unit": null}, "slug": "llm-stats-cbnsl", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CC-Bench-V2 Backend evaluates coding agents on backend development tasks, measuring practical engineering ability to implement server-side logic, APIs, and system components.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cc-bench-v2-backend:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cc-bench-v2-backend", "languages": [], "modality": "text", "name": "CC-Bench-V2 Backend", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 22.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.228, "raw_min": 0.228, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cc-bench-v2-backend:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500"}, "unit": null}, "slug": "llm-stats-cc-bench-v2-backend", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CC-Bench-V2 Frontend evaluates coding agents on frontend development tasks, measuring ability to build UI components, handle styling, and implement client-side logic.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cc-bench-v2-frontend:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cc-bench-v2-frontend", "languages": [], "modality": "text", "name": "CC-Bench-V2 Frontend", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.684, "raw_min": 0.684, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cc-bench-v2-frontend:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500"}, "unit": null}, "slug": "llm-stats-cc-bench-v2-frontend", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CC-Bench-V2 Repo Exploration evaluates coding agents on repository-level understanding and navigation, measuring ability to explore, comprehend, and work across entire codebases.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cc-bench-v2-repo:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cc-bench-v2-repo", "languages": [], "modality": "text", "name": "CC-Bench-V2 Repo Exploration", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.2, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.722, "raw_min": 0.722, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cc-bench-v2-repo:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500"}, "unit": null}, "slug": "llm-stats-cc-bench-v2-repo", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "structured_output", "text-to-image", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive OCR benchmark for evaluating Large Multimodal Models (LMMs) in literacy. Comprises four OCR-centric tracks: multi-scene text reading, multilingual text reading, document parsing, and key information extraction. Contains 39 subsets with 7,058 fully annotated images, 41% sourced from real applications. Tests capabilities including text grounding, multi-orientation text recognition, and detecting hallucination/repetition across diverse visual challenges.", "evidence_summary": {"document_count": 1, "model_count": 18, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cc-ocr:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cc-ocr", "languages": [], "modality": "multimodal", "name": "CC-OCR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 18, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.39999999999999, "display_multiplier": 100, "model_count": 18, "model_count_basis": "source_model_id", "numeric_count": 18, "raw_max": 0.834, "raw_min": 0.738, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cc-ocr:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500"}, "unit": null}, "slug": "llm-stats-cc-ocr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CEB evaluates LLM bias compositionally, featuring 11k samples characterized across bias types, social groups, and tasks. CEB是一个用于大型语言模型偏差的组成评估基准，引入了包含 11,004 个样本的组成评估基准，从偏差类型、社会群体和任务三个维度描述每个数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1552", "languages": [], "modality": null, "name": "CEB", "openness": "restricted", "publisher": "University of Virginia, Arizona State University, etc.", "released": "2024-07-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1552-ceb", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CEB", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CFEval benchmark for evaluating code generation and problem-solving capabilities", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cfeval:qwen3-235b-a22b-thinking-2507", "reported_at": "2025-07-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cfeval", "languages": [], "modality": "text", "name": "CFEval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 2134.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 2134.0, "raw_min": 2071.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cfeval:qwen3-235b-a22b-thinking-2507", "reported_date": "2025-07-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500"}, "unit": null}, "slug": "llm-stats-cfeval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "NAACL 2025", "金融", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "We present CFinBench: a meticulously crafted, the most comprehensive evaluation benchmark to date, for assessing the financial knowledge of LLMs under Chinese context. 为了更加全面地探究大语言模型在中文财经领域的能力，本工作提出了目前为止量级最大的中文财经评测基准（CFinBench）。该数据集共包含99,100个评测样本，并包含单选题、多选题和判断题在内的三种题型。该工作对当前主流的大模型从四个维度进行了详细评测：财经学科基础、财经资格认证、财经从业实践、财经法律法规。数据集和测评代码均已开源。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": true, "key": "opencompass:1907", "languages": ["Chinese"], "modality": null, "name": "CFinBench", "openness": "unknown", "publisher": "华为, 新加坡南洋理工", "released": "2024-10-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1907-cfinbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CFinBench", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "ACL 2024", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CFLUE is the Chinese Financial Language Understanding Evaluation benchmark, designed to assess the capability of LLMs across various dimensions. CFLUE 是中国金融语言理解评估基准，旨在评估大型语言模型（LLMs）在各个维度上的能力。具体而言，CFLUE 提供了针对知识评估和应用评估量身定制的数据集。在知识评估方面，它包含超过 38,000 道选择题及相关的解决方案解释。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1079", "languages": [], "modality": null, "name": "CFLUE", "openness": "unknown", "publisher": "Alibaba", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1079-cflue", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CFLUE", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CG-Bench is meant for evaluating MLLMs' long video understanding, including 12,129 QA pairs from 1219 videos in 3 major question types: perception, reasoning, and hallucination. CG-Bench用于评估多模态大模型的长视频理解能力，基于1219个视频设计了12129个涵盖感知、推理和幻觉三种问题类型的QA对。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1513", "languages": [], "modality": "multimodal", "name": "CG-Bench", "openness": "restricted", "publisher": "Nanjing University", "released": "2024-12-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1513-cg-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CG-Bench", "unit": null}, {"aliases": [], "categories": ["multimodal", "language", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Charades-STA is a benchmark dataset for temporal activity localization via language queries, extending the Charades dataset with sentence temporal annotations. It contains 12,408 training and 3,720 testing segment-sentence pairs from videos with natural language descriptions and precise temporal boundaries for localizing activities based on language queries.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:charadessta:qwen2.5-vl-7b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:charadessta", "languages": [], "modality": "multimodal", "name": "CharadesSTA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.8, "display_multiplier": 100, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 0.648, "raw_min": 0.436, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:charadessta:qwen3-vl-235b-a22b-instruct", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500"}, "unit": null}, "slug": "llm-stats-charadessta", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "ACL 2024", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CHARM is the first benchmark for comprehensively and in-depth evaluating the commonsense reasoning ability of large language models (LLMs) in Chinese, which covers both globally known and Chinese-specific commonsense. CHARM 是首个全面深入评估大语言模型（LLMs）在中文中的常识推理能力的基准，涵盖了全球通用的常识和特定于中国的常识。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1141", "languages": ["Chinese"], "modality": null, "name": "CHARM", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1141-charm", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CHARM", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ChartMuseum is a chart question-answering benchmark of 1,162 expert-annotated questions over real-world chart images drawn from 184 sources, including academic figures, infographics, and unconventional chart designs. It specifically targets questions that require visual reasoning, such as comparing unlabeled visual elements, tracking trajectories, and judging spatial relationships.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:chartmuseum:claude-sonnet-5", "reported_at": "2026-06-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:chartmuseum", "languages": [], "modality": "multimodal", "name": "ChartMuseum", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.867, "raw_min": 0.867, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:chartmuseum:claude-sonnet-5", "reported_date": "2026-06-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500"}, "unit": null}, "slug": "llm-stats-chartmuseum", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500", "unit": null}, {"aliases": ["Chartography"], "categories": ["vision"], "collected_at": null, "description": "Chart comprehension benchmark with tool access. Scores depend on context length and tool configuration.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000chartography\u0000zai_glm_5_3_flash_model_card\u0000chartography\u0000Chartography, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000chartography\u0000zai_glm_5_3_flash_model_card\u0000chartography\u0000Chartography, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:chartography", "languages": [], "modality": null, "name": "Chartography", "openness": "unknown", "publisher": null, "released": "2025-06-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:chartography", "source_url": "https://github.com/Chartography/Chartography"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.0, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 78.0, "raw_min": 78.0, "source_reference": {"obs_id": "curated\u0000chartography\u0000zai_glm_5_3_flash_model_card\u0000chartography\u0000Chartography, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "observation_id": "curated\u0000chartography\u0000zai_glm_5_3_flash_model_card\u0000chartography\u0000Chartography, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "chartography", "source": "model_reports", "source_url": "https://github.com/Chartography/Chartography", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ChartQA is a large-scale benchmark comprising 9.6K human-written questions and 23.1K questions generated from human-written chart summaries, designed to evaluate models' abilities in visual and logical reasoning over charts.", "evidence_summary": {"document_count": 1, "model_count": 26, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:chartqa:grok-1.5v", "reported_at": "2024-04-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:chartqa", "languages": [], "modality": "multimodal", "name": "ChartQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 26, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.8, "display_multiplier": 100, "model_count": 26, "model_count_basis": "source_model_id", "numeric_count": 26, "raw_max": 0.908, "raw_min": 0.688, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:chartqa:claude-3-5-sonnet-20241022", "reported_date": "2024-10-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500"}, "unit": null}, "slug": "llm-stats-chartqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500", "unit": null}, {"aliases": ["ChartQA"], "categories": ["multimodal"], "collected_at": null, "description": "Largely saturated; relaxed-accuracy tolerance affects the reported figure.", "evidence_summary": {"document_count": 2, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:chartqa", "languages": [], "modality": null, "name": "ChartQA", "openness": "unknown", "publisher": null, "released": "2022-03-19", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:chartqa", "source_url": "https://github.com/vis-nlp/ChartQA"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "chartqa", "source": "model_reports", "source_url": "https://github.com/vis-nlp/ChartQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ChartQAPro is a challenging benchmark for question answering over diverse, real-world charts and infographics.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:chartqapro:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:chartqapro", "languages": [], "modality": "multimodal", "name": "ChartQAPro", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.89999999999999, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.709, "raw_min": 0.709, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:chartqapro:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500"}, "unit": null}, "slug": "llm-stats-chartqapro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500", "unit": null}, {"aliases": ["CharXiv Reasoning", "CharXiv"], "categories": ["vision"], "collected_at": null, "description": "Chart reasoning benchmark with tool access. Scores depend on context length and tool configuration.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000charxiv_reasoning\u0000zai_glm_5_3_flash_model_card\u0000charxiv_reasoning\u0000CharXiv Reasoning, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000charxiv_reasoning\u0000zai_glm_5_3_flash_model_card\u0000charxiv_reasoning\u0000CharXiv Reasoning, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:charxiv_reasoning", "languages": [], "modality": null, "name": "CharXiv Reasoning", "openness": "unknown", "publisher": null, "released": "2025-06-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:charxiv_reasoning", "source_url": "https://github.com/CharXiv/CharXiv"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 89.4, "raw_min": 89.4, "source_reference": {"obs_id": "curated\u0000charxiv_reasoning\u0000zai_glm_5_3_flash_model_card\u0000charxiv_reasoning\u0000CharXiv Reasoning, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "observation_id": "curated\u0000charxiv_reasoning\u0000zai_glm_5_3_flash_model_card\u0000charxiv_reasoning\u0000CharXiv Reasoning, w/ tools, temp 1.0, top_p 0.95, max context 256K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "charxiv_reasoning", "source": "model_reports", "source_url": "https://github.com/CharXiv/CharXiv", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "structured_output", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CharXiv-D is the descriptive questions subset of the CharXiv benchmark, designed to assess multimodal large language models' ability to extract basic information from scientific charts. It contains descriptive questions covering information extraction, enumeration, pattern recognition, and counting across 2,323 diverse charts from arXiv papers, all curated and verified by human experts.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:charxiv-d:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:charxiv-d", "languages": [], "modality": "multimodal", "name": "CharXiv-D", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.5, "display_multiplier": 100, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 0.955, "raw_min": 0.6, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:charxiv-d:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500"}, "unit": null}, "slug": "llm-stats-charxiv-d", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CharXiv-R is the reasoning component of the CharXiv benchmark, focusing on complex reasoning questions that require synthesizing information across visual chart elements. It evaluates multimodal large language models on their ability to understand and reason about scientific charts from arXiv papers through various reasoning tasks.", "evidence_summary": {"document_count": 1, "model_count": 51, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:charxiv-r:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:charxiv-r", "languages": [], "modality": "multimodal", "name": "CharXiv-R", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 51, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.2, "display_multiplier": 100, "model_count": 51, "model_count_basis": "source_model_id", "numeric_count": 51, "raw_max": 0.932, "raw_min": 0.397, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:charxiv-r:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500"}, "unit": null}, "slug": "llm-stats-charxiv-r", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "数学", "Math", "大语言模型", "LLM", "代码工程", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CHASE is a unified framework to synthetically generate challenging problems using LLMs without human involvement CHASE是一个无需人工参与的统一框架，用于合成生成具有挑战性的问题", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1534", "languages": [], "modality": null, "name": "CHASE-Code", "openness": "open", "publisher": "Mila and McGill University", "released": "2025-02-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1534-chase-code", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CHASE-Code", "unit": null}, {"aliases": ["Chatbot Arena", "LMArena", "LMSYS Arena"], "categories": ["human_preference"], "collected_at": null, "description": "Elo from real user traffic; sampling and prompt distribution are outside any vendor's control.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:chatbot_arena", "languages": [], "modality": null, "name": "Chatbot Arena", "openness": "unknown", "publisher": null, "released": "2023-05-03", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:chatbot_arena", "source_url": "https://lmarena.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "chatbot_arena", "source": "model_reports", "source_url": "https://lmarena.ai/", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "科学智能", "AI for Science", "知识储备", "科学推理", "Scientific Reasoning", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ChemBench is a large-scale chemistry competency evaluation benchmark for language models, which includes nine chemistry core tasks and 4100 high-quality single-choice questions and answers. ChemBench是一个包含了九项化学核心任务，4100个高质量单选问答的大语言模型化学能力评测基准.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:692", "languages": [], "modality": null, "name": "ChemBench", "openness": "unknown", "publisher": null, "released": "2024-02-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-692-chembench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ChemBench", "unit": null}, {"aliases": [], "categories": ["healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CheXpert is a large dataset of 224,316 chest radiographs from 65,240 patients for automated chest X-ray interpretation. The dataset includes uncertainty labels for 14 medical observations extracted from radiology reports. It serves as a benchmark for developing and evaluating automated chest radiograph interpretation models.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:chexpert-cxr:medgemma-4b-it", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:chexpert-cxr", "languages": [], "modality": "image", "name": "CheXpert CXR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.481, "raw_min": 0.481, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:chexpert-cxr:medgemma-4b-it", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500"}, "unit": null}, "slug": "llm-stats-chexpert-cxr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CHID is a chinese idiom reading comprehension task, which requires to select the correct idiom to fill in the blank according to the context, with 10 candidate idioms. CHID是一个中文成语阅读理解任务，要求根据上下文选择正确的成语填空，共有10个候选成语。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:505", "languages": ["Chinese"], "modality": null, "name": "CHID", "openness": "unknown", "publisher": null, "released": "2019-06-04", "released_reference": {"basis": "paper_first_version", "note": "First version of the paper introducing the Chinese idiom cloze dataset.", "source_key": "opencompass:505", "source_url": "https://arxiv.org/abs/1906.01265"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-505-chid", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CHID", "unit": null}, {"aliases": [], "categories": ["创作", "Creation", "NeurIPS 2024", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ChronoMagic-Bench can evaluate the temporal and metamorphic capabilities of the T2V (text-to-video) models in time-lapse video generation, introducing 1,649 prompts and real-world videos as references. ChronoMagic-Bench用来评估 T2V （文本到视频 ）模型在延时视频生成中的时间和变形能力，引入了1649个提示和真实世界的视频作为参考。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1278", "languages": [], "modality": null, "name": "ChronoMagic-Bench", "openness": "open", "publisher": "Peking University", "released": "2024-06-26", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1278-chronomagic-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ChronoMagic-Bench", "unit": null}, {"aliases": [], "categories": ["memory", "privacy", "safety", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CI Memories measures privacy behavior in memory-enabled agents using contextual-integrity scenarios. This metric reports evaluation coverage.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ci-memories-coverage:muse-glimmer-30b", "reported_at": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ci-memories-coverage", "languages": [], "modality": "text", "name": "CI Memories Coverage", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.648, "raw_min": 0.648, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ci-memories-coverage:muse-glimmer-30b", "reported_date": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500"}, "unit": null}, "slug": "llm-stats-ci-memories-coverage", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500", "unit": null}, {"aliases": [], "categories": ["memory", "privacy", "safety", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CI Memories measures privacy failures in memory-enabled agents using contextual-integrity scenarios. This metric is the violation rate; lower is better.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ci-memories-violation:muse-glimmer-30b", "reported_at": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ci-memories-violation", "languages": [], "modality": "text", "name": "CI Memories Violation Rate", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 26.400000000000002, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.264, "raw_min": 0.264, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ci-memories-violation:muse-glimmer-30b", "reported_date": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500"}, "unit": null}, "slug": "llm-stats-ci-memories-violation", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CL-bench is an open-source benchmark with its own data and rubrics for evaluating models on coding and agentic tasks, scored using a setup fully aligned with the official procedure.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cl-bench:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cl-bench", "languages": [], "modality": "text", "name": "CL-bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 23.799999999999997, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.238, "raw_min": 0.2048, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cl-bench:hy3", "reported_date": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-cl-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CL-bench Life variant.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cl-bench-(life):hy3", "reported_at": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cl-bench-(life)", "languages": [], "modality": "text", "name": "CL-bench (Life)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 17.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.17, "raw_min": 0.17, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cl-bench-(life):hy3", "reported_date": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500"}, "unit": null}, "slug": "llm-stats-cl-bench-life", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Claw-Eval tests real-world agentic task completion across complex multi-step scenarios, evaluating a model's ability to use tools, navigate environments, and complete end-to-end tasks autonomously.", "evidence_summary": {"document_count": 1, "model_count": 14, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:claw-eval:mimo-v2-omni", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:claw-eval", "languages": [], "modality": "text", "name": "Claw-Eval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 14, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.9, "display_multiplier": 100, "model_count": 14, "model_count_basis": "source_model_id", "numeric_count": 14, "raw_max": 0.809, "raw_min": 0.5, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:claw-eval:kimi-k2.6", "reported_date": "2026-04-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500"}, "unit": null}, "slug": "llm-stats-claw-eval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ClawEval-MM is the multimodal variant of ClawEval, evaluating agentic problem solving with visual inputs.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:claw-eval-mm:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:claw-eval-mm", "languages": [], "modality": "multimodal", "name": "ClawEval-MM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.89999999999999, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.569, "raw_min": 0.46, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:claw-eval-mm:qwen3.8-27b", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500"}, "unit": null}, "slug": "llm-stats-claw-eval-mm", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "安全", "Safety", "大语言模型", "LLM", "语言理解", "Comprehension", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CleanPatrick is a large-scale, real-world image data-cleaning benchmark with 496,377 binary annotations from 933 medical crowd workers for ranking off-topic, near-duplicate, and label-error issues. CleanPatrick是首个大规模图像数据清洗基准，基于公开的Fitzpatrick17k皮肤科数据集构建。该基准包含超过50万条来自933名医学众包工人的二元注释，涵盖三种数据质量问题：离题样本、近似重复样本和标签错误。通过医学专家验证，CleanPatrick提供了高质量的基准数据，用于评估图像数据清洗策略。基准测试结果表明，现有的数据清洗方法在近似重复检测中表现出色，但在标签错误检测方面仍面临挑战。CleanPatrick为数据清洗方法提供了标准化的评估框架，推动了更可靠的数据中心人工智能的发展。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1849", "languages": [], "modality": null, "name": "CleanPatrick", "openness": "restricted", "publisher": "University of Basel,etc", "released": "2025-05-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1849-cleanpatrick", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CleanPatrick", "unit": null}, {"aliases": [], "categories": ["创作", "Creation", "大语言模型", "LLM", "语言生成", "Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "clembench is a benchmark framework for evaluating LLMs through dialogue game-based interactions, assessing chat-optimized models as conversational agents via multi-turn interactive game scenario. clembench是一个基于对话游戏的大语言模型评测框架，通过多回合交互式游戏场景评测聊天优化模型的会话智能体能力，包含Wordle、Taboo等多种游戏任务，评测指令遵循、目标导向行为和精细化交互理解等核心维度，提供可重复、可控制的自玩评测环境和综合得分机制，支持英语及多语言扩展的开源评测平台。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2127", "languages": [], "modality": null, "name": "clembench", "openness": "unknown", "publisher": "Computational Linguistics ,etc", "released": "2025-07-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2127-clembench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/clembench", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "科学智能", "AI for Science", "代码工程", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CLEVER is a benchmark suite for end-to-end code generation and formal verification in Lean 4, adapted from the HumanEval dataset. It requires models to generate implementations, formal specifications, and proofs—all verifiable by Lean's type checker, moving beyond test-case-driven evaluation. CLEVER is a benchmark suite for end-to-end code generation and formal verification in Lean 4, adapted from the HumanEval dataset. It requires models to generate implementations, formal specifications, and proofs—all verifiable by Lean's type checker, moving beyond test-case-driven evaluation.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1834", "languages": [], "modality": null, "name": "CLEVER", "openness": "open", "publisher": "University of Texas at Austin ,etc", "released": "2025-05-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1834-clever", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CLEVER", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "知识", "Knowledge", "Multimodal Reasoning", "Fact-Checking", "Charts", "科学智能", "AI for Science", "逻辑推理", "知识储备", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ClimateViz is a large-scale multimodal benchmark designed to evaluate the scientific fact-checking and statistical reasoning capabilities of large language and vision-language models. It focuses on real-world climate science data, with over 49,000 high quality natural language claims. ClimateViz 是一个多模态基准数据集，用于评估大模型在气候科学图表上的事实核查与统计推理能力。数据来源于 NOAA、英国气象局等权威机构，共包含约 2,800 张科学图表与近 5 万条主张，标注为支持（support）、反驳（refute）或信息不足（NEI）。\n\n该数据集支持三种输入格式：图表+主张、表格+主张、图表标题+表格+主张，涵盖趋势识别、时空推理与科学对比等核心任务。ClimateViz 适用于多模态大模型和语言模型的系统性评估。\n图表转表格 + 图表标题 + 主张（Caption + Table", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1954", "languages": [], "modality": "multimodal", "name": "ClimateViz", "openness": "unknown", "publisher": "University of Oxford", "released": "2025-06-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1954-climateviz", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ClimateViz", "unit": null}, {"aliases": [], "categories": ["reasoning", "safety", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CloningScenarios is an expert-level multi-step reasoning benchmark about difficult genetic cloning scenarios in multiple-choice format. It evaluates dual-use biological knowledge relevant to bioweapons development.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cloningscenarios:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cloningscenarios", "languages": [], "modality": "text", "name": "CloningScenarios", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 46.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.46, "raw_min": 0.46, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cloningscenarios:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500"}, "unit": null}, "slug": "llm-stats-cloningscenarios", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CLUEWSC2020 is the Chinese version of the Winograd Schema Challenge, part of the CLUE benchmark. It focuses on pronoun disambiguation and coreference resolution, requiring models to determine which noun a pronoun refers to in a sentence. The dataset contains 1,244 training samples and 304 development samples extracted from contemporary Chinese literature.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cluewsc:deepseek-v3", "reported_at": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cluewsc", "languages": [], "modality": "text", "name": "CLUEWSC", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.4, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.914, "raw_min": 0.486, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cluewsc:kimi-k1.5", "reported_date": "2025-01-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500"}, "unit": null}, "slug": "llm-stats-cluewsc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "NAACL 2024", "大语言模型", "LLM", "知识储备", "Knowledge", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CMB is Comprehensive Medical Benchmark in Chinese, designed and rooted entirely within the native Chinese linguistic and cultural framework. CMB 是一个综合医学基准，专为中文而设计，并完全依赖于本土的中文语言和文化框架中。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1137", "languages": ["English", "Chinese"], "modality": null, "name": "CMB", "openness": "open", "publisher": "The Chinese University of Hong Kong & Shenzhen Research Institute of Big Data", "released": "2024-04-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1137-cmb", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CMB", "unit": null}, {"aliases": [], "categories": ["reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CMMLU (Chinese Massive Multitask Language Understanding) is a comprehensive Chinese benchmark that evaluates the knowledge and reasoning capabilities of large language models across 67 different subject topics. The benchmark covers natural sciences, social sciences, engineering, and humanities with multiple-choice questions ranging from basic to advanced professional levels.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cmmlu:qwen2-72b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cmmlu", "languages": [], "modality": "text", "name": "CMMLU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.2, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.902, "raw_min": 0.398, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cmmlu:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500"}, "unit": null}, "slug": "llm-stats-cmmlu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CMMLU is a comprehensive Chinese evaluation benchmark specifically designed to assess the knowledge and reasoning abilities of language models in the context of the Chinese language. CMMLU covers 67 topics ranging from basic subjects to advanced professional levels. It includes tasks that require calculations and reasoning in natural sciences, as well as tasks involving knowledge from humanities, social sciences, and practical aspects like Chinese driving rules. Moreover, many tasks within CMMLU have answers specific to China, which might not be universally applicable in other regions or languages. As a result, CMMLU serves as a fully localized Chinese evaluation benchmark. CMMLU是一个综合性的中文评估基准，专门用于评估语言模型在中文语境下的知识和推理能力。CMMLU涵盖了从基础学科到高级专业水平的67个主题。它包括：需要计算和推理的自然科学，需要知识的人文科学和社会科学,以及需要生活常识的中国驾驶规则等。此外，CMMLU中的许多任务具有中国特定的答案，可能在其他地区或语言中并不普遍适用。因此是一个完全中国化的中文测试基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:499", "languages": ["Chinese"], "modality": null, "name": "CMMLU", "openness": "unknown", "publisher": null, "released": "2023-06-15", "released_reference": {"basis": "paper_first_version", "note": "First version introducing CMMLU; the 2024 revision is not its release.", "source_key": "opencompass:499", "source_url": "https://arxiv.org/abs/2306.09212"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-499-cmmlu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CMMLU", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CMNLI  is a Chinese natural language inference task, which requires to determine the logical relation between two sentences, with three relations: entailment, contradiction and neutral. CMNLI是一个中文自然语言推理任务，要求根据两个句子判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:524", "languages": ["Chinese"], "modality": null, "name": "CMNLI", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-524-cmnli", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CMNLI", "unit": null}, {"aliases": [], "categories": ["physics", "reasoning", "science"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CMT-Benchmark evaluates models on condensed matter theory problems, testing advanced physics reasoning across areas such as many-body systems, quantum field theory, and statistical mechanics.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cmt-benchmark:hy3", "reported_at": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cmt-benchmark", "languages": [], "modality": "text", "name": "CMT-Benchmark", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 37.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.379, "raw_min": 0.379, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cmt-benchmark:hy3", "reported_date": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500"}, "unit": null}, "slug": "llm-stats-cmt-benchmark", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500", "unit": null}, {"aliases": [], "categories": ["math"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "China Mathematical Olympiad 2024 - A challenging mathematics competition.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cnmo-2024:deepseek-v3", "reported_at": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cnmo-2024", "languages": [], "modality": "text", "name": "CNMO 2024", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.3, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.743, "raw_min": 0.432, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cnmo-2024:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500"}, "unit": null}, "slug": "llm-stats-cnmo-2024", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CodeCriticBench assesses LLMs' critiquing ability in code generation and QA tasks. Covering 10 criteria, it features a 4.3k-samples dataset with three difficulty levels and balanced distribution. CodeCriticBench ，旨在系统地评估LLMs在代码生成和代码问答任务中的批评能力。其涵盖 10 个不同的标准，数据集根据难度分为三个等级，共包含4.3k个样本，确保了难度级别的平衡分布。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1546", "languages": [], "modality": null, "name": "CodeCriticBench", "openness": "restricted", "publisher": "NJU, M-A-P, Alibaba, etc.", "released": "2025-02-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1546-codecriticbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CodeCriticBench", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Researchers introduce CodeElo, a standardized competition-level code generation benchmark that effectively addresses all these challenges for the first time. CodeElo benchmark is mainly based on the official CodeForces platform and tries to align with the platform as much as possible. CodeElo，这是一个标准化的竞赛级代码生成基准测试，有效解决了所有这些挑战。CodeElo 基准测试主要基于官方 CodeForces 平台，并尽可能与该平台保持一致。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1624", "languages": [], "modality": null, "name": "CodeElo", "openness": "open", "publisher": "QwenTeam, Alibaba Group", "released": "2025-01-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1624-codeelo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CodeElo", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A competitive programming benchmark using problems from the CodeForces platform. The benchmark evaluates code generation capabilities of LLMs on algorithmic problems with difficulty ratings ranging from 800 to 2400. Problems cover diverse algorithmic categories including dynamic programming, graph algorithms, data structures, and mathematical problems with standardized evaluation through direct platform submission.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:codeforces:deepseek-v3.1", "reported_at": "2025-01-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:codeforces", "languages": [], "modality": "text", "name": "CodeForces", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 1.0, "display_multiplier": 1, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 1.0, "raw_min": 0.4763, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:codeforces:deepseek-v4-flash-max", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500"}, "unit": null}, "slug": "llm-stats-codeforces", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500", "unit": null}, {"aliases": ["Codeforces", "Codeforces ELO", "Codeforces Rating"], "categories": ["coding"], "collected_at": null, "description": "An Elo estimate against human contestants, not a fixed dataset. The problem set moves continuously and rating conversion differs by vendor.", "evidence_summary": {"document_count": 3, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000codeforces\u0000google_gemma_4_model_card\u0000codeforces\u0000Elo\u0000Gemma 4 (31B)", "reported_at": "2026-03-11", "source_url": "https://huggingface.co/google/gemma-4-31B-it"}, "first_score_reported_at": "2026-03-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000codeforces\u0000google_gemma_4_model_card\u0000codeforces\u0000Elo\u0000Gemma 4 (31B)", "reported_at": "2026-03-11", "source_url": "https://huggingface.co/google/gemma-4-31B-it"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:codeforces", "languages": [], "modality": null, "name": "Codeforces", "openness": "unknown", "publisher": null, "released": "2010-02-19", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:codeforces", "source_url": "https://codeforces.com/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 3206.0, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 3206.0, "raw_min": 2150.0, "source_reference": {"obs_id": "curated\u0000codeforces\u0000deepseek_v4_model_card\u0000codeforces\u0000think max, rating\u0000DeepSeek-V4-Pro", "observation_id": "curated\u0000codeforces\u0000deepseek_v4_model_card\u0000codeforces\u0000think max, rating\u0000DeepSeek-V4-Pro", "reported_at": "2026-04-22", "reported_date": "2026-04-22", "source_id": "deepseek_v4_model_card", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"}, "unit": "elo"}, "slug": "codeforces", "source": "model_reports", "source_url": "https://codeforces.com/", "unit": "elo"}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Codegolf v2.2 benchmark", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:codegolf-v2.2:gemma-3n-e2b-it-litert-preview", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:codegolf-v2.2", "languages": [], "modality": "text", "name": "Codegolf v2.2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 16.8, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.168, "raw_min": 0.11, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:codegolf-v2.2:gemma-3n-e4b-it", "reported_date": "2025-06-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500"}, "unit": null}, "slug": "llm-stats-codegolf-v2-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CodeMMLU is a comprehensive benchmark designed to evaluate the capabilities of large language models (LLMs) in coding and software knowledge. It builds upon the structure of multiple-choice question answering (MCQA) to cover a wide range of programming tasks and domains. CodeMMLU 是一个旨在评估大型语言模型（LLMs）在编码和软件知识方面能力的全面基准。它基于多项选择题回答（MCQA）的结构，涵盖了广泛的编程任务和领域，包括代码生成、缺陷检测、软件工程原则等。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1582", "languages": [], "modality": null, "name": "CodeMMLU", "openness": "open", "publisher": "Monash University, CSIRO’s Data61, etc.", "released": "2024-06-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1582-codemmlu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CodeMMLU", "unit": null}, {"aliases": [], "categories": ["question_answering", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Cohere's internal North evaluation for measuring how well a model answers enterprise questions using MCP-connected cloud file systems. Scores are reported with LLM-as-a-judge techniques.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cohere-agentic-question-answering:command-a-plus-05-2026", "reported_at": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cohere-agentic-question-answering", "languages": [], "modality": "text", "name": "Cohere Agentic Question Answering", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.65, "raw_min": 0.65, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cohere-agentic-question-answering:command-a-plus-05-2026", "reported_date": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500"}, "unit": null}, "slug": "llm-stats-cohere-agentic-question-answering", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "data_analysis", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Cohere's internal North evaluation for measuring a model's ability to perform data science tasks over uploaded spreadsheets. Scores are reported with LLM-as-a-judge techniques.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cohere-data-analysis:command-a-plus-05-2026", "reported_at": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cohere-data-analysis", "languages": [], "modality": "text", "name": "Cohere Data Analysis", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 45.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.45, "raw_min": 0.45, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cohere-data-analysis:command-a-plus-05-2026", "reported_date": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500"}, "unit": null}, "slug": "llm-stats-cohere-data-analysis", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500", "unit": null}, {"aliases": [], "categories": ["memory", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Cohere's internal North evaluation for measuring how well an agent uses information from North's memory system across sessions. Scores are reported with LLM-as-a-judge techniques.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cohere-memory-usage-quality:command-a-plus-05-2026", "reported_at": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cohere-memory-usage-quality", "languages": [], "modality": "text", "name": "Cohere Memory Usage Quality", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 54.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.54, "raw_min": 0.54, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cohere-memory-usage-quality:command-a-plus-05-2026", "reported_date": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500"}, "unit": null}, "slug": "llm-stats-cohere-memory-usage-quality", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "COLLIE is a grammar-based framework for systematic construction of constrained text generation tasks. It allows specification of rich, compositional constraints across diverse generation levels and modeling challenges including language understanding, logical reasoning, and semantic planning. The COLLIE-v1 dataset contains 2,080 instances across 13 constraint structures.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:collie:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:collie", "languages": [], "modality": "text", "name": "COLLIE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.0, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.99, "raw_min": 0.425, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:collie:gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500"}, "unit": null}, "slug": "llm-stats-collie", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "大语言模型", "LLM", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "COLLIE is a grammar-based framework that allows the specification of rich, compositional constraints with diverse generation levels (word, sentence, paragraph, passage) and modeling challenges (e.g.,language understanding, logical reasoning, counting, semantic planning). COLLIE用于评估大模型在约束性文本生成任务中的表现，可指定具有不同生成级别（单词、句子、段落、段落）和建模挑战（例如，语言理解、逻辑推理、计数、语义规划）的丰富组合。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1250", "languages": [], "modality": null, "name": "Collie", "openness": "unknown", "publisher": "Princeton University", "released": "2023-07-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1250-collie", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Collie", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ColorBench, an innovative benchmark meticulously crafted to assess the capabilities of VLMs in color understanding, including color perception, reasoning, and robustness. ColorBench，这是一个创新且精心设计的基准测试，旨在评估VLMs在颜色理解方面的能力，包括颜色感知、推理和鲁棒性。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1777", "languages": [], "modality": "multimodal", "name": "ColorBench", "openness": "unknown", "publisher": "University of Maryland, College Park", "released": "2025-04-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1777-colorbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ColorBench", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "Combinatorics", "Lean4", "科学智能", "AI for Science", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "We introduce CombiBench, a comprehensive benchmark comprising 100 combinatorial problems, each formalized in Lean4 and paired with its corresponding informal statement. The problem set covers a wide spectrum of difficulty levels, ranging from middle school to IMO and university level. CombiBench是一个包含100个组合问题的综合基准测试，每个问题都用Lean4进行了形式化，并附有其对应的非形式化表述。这些问题涵盖了从中学生到国际数学奥林匹克竞赛（IMO）以及大学水平的广泛难度范围。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1943", "languages": [], "modality": null, "name": "CombiBench", "openness": "unknown", "publisher": "AMSS, SYSU, Moonshot AI, Numina", "released": "2025-04-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1943-combibench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CombiBench", "unit": null}, {"aliases": [], "categories": ["speech_to_text", "language", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Common Voice is a massively-multilingual collection of transcribed speech intended for speech technology research and development. Version 15.0 contains 28,750 recorded hours across 114 languages, consisting of crowdsourced voice recordings with corresponding transcriptions.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:common-voice-15:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:common-voice-15", "languages": [], "modality": "audio", "name": "Common Voice 15", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.076, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.076, "raw_min": 0.076, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:common-voice-15:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500"}, "unit": null}, "slug": "llm-stats-common-voice-15", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CommonSenseQA is a multiple-choice question answering dataset that requires different types of commonsense knowledge to predict correct answers. It contains 12,102 questions with one correct answer and four distractors, designed to test semantic reasoning and conceptual relationships. Questions are created based on ConceptNet concepts and require prior world knowledge for accurate reasoning.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:commonsenseqa:mistral-nemo-instruct-2407", "reported_at": "2024-07-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:commonsenseqa", "languages": [], "modality": "text", "name": "CommonSenseQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.39999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.704, "raw_min": 0.704, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:commonsenseqa:mistral-nemo-instruct-2407", "reported_date": "2024-07-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500"}, "unit": null}, "slug": "llm-stats-commonsenseqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CommonsenseQA is a multiple-choice question answering dataset that requires different types of commonsense knowledge to predict the correct answers . It contains 12,102 questions with one correct answer and four distractor answers. CommonsenseQA是一个选择题数据集，它需要不同类型的常识知识来预测正确答案。它包含12,102个问题，有一个正确答案和四个干扰答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:511", "languages": [], "modality": null, "name": "CommonSenseQA", "openness": "unknown", "publisher": null, "released": "2018-11-02", "released_reference": {"basis": "paper_first_version", "note": "First version of the CommonsenseQA dataset paper.", "source_key": "opencompass:511", "source_url": "https://arxiv.org/abs/1811.00937"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-511-commonsenseqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CommonSenseQA", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "NeurIPS 2024", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CompBench is designed to evaluate the comparative reasoning capability of multimodal large language models, including a collection of around 40K image pairs and visually oriented questions covering 8 dimensions CompBench旨在评估多模态大模型的比较推理能力，包含约4万个图像对及8个维度的视觉配对问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1336", "languages": [], "modality": "multimodal", "name": "CompBench", "openness": "unknown", "publisher": "The Ohio State University", "released": "2024-07-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1336-compbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CompBench", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "structured_output", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ComplexFuncBench is a benchmark designed to evaluate large language models' capabilities in handling complex function calling scenarios. It encompasses multi-step and constrained function calling tasks that require long-parameter filling, parameter value reasoning, and managing contexts up to 128k tokens. The benchmark includes 1,000 samples across five real-world scenarios.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:complexfuncbench:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:complexfuncbench", "languages": [], "modality": "text", "name": "ComplexFuncBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.5, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.665, "raw_min": 0.057, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:complexfuncbench:gpt-4o-2024-08-06", "reported_date": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500"}, "unit": null}, "slug": "llm-stats-complexfuncbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Connectors is an OpenAI internal production benchmark measuring reliable use of connector-based tools in agentic workflows, reported as a pass rate.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openai-connectors:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openai-connectors", "languages": [], "modality": "text", "name": "Connectors", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 100.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 1.0, "raw_min": 0.999, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openai-connectors:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500"}, "unit": null}, "slug": "llm-stats-openai-connectors", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ContextualJudgeBench is a pairwise benchmark with 2,000 samples for evaluating LLM-as-judge models in two contextual settings: Contextual QA and summarization. We propose a pairwise evaluation hierarchy and generate splits for our proposed hierarchy. ContextualJudgeBench 是一个包含 2,000 个样本的成对基准，用于评估在两个上下文环境下的LLM-as-judge 模型：上下文问答和摘要。我们提出一个成对评估层次结构，并为我们的层次结构生成分割。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1679", "languages": [], "modality": null, "name": "ContextualJudgeBench", "openness": "open", "publisher": "Salesforce AI Research", "released": "2025-03-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1679-contextualjudgebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ContextualJudgeBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ContPhy is a continuum physical-reasoning benchmark evaluating understanding of physical dynamics in video.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:contphy:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:contphy", "languages": [], "modality": "multimodal", "name": "ContPhy", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.636, "raw_min": 0.611, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:contphy:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500"}, "unit": null}, "slug": "llm-stats-contphy", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ConvBench is a novel multi-turn conversation evaluation benchmark tailored for Large Vision-Language Models (LVLMs). It comprises 577 meticulously curated multi-turn conversations encompassing 215 tasks reflective of real-world demands. ConvBench是专用于大型视觉语言模型 （LVLM）的新型多轮对话评估基准，由基于215 个反映实际需求的任务的577个多轮次对话组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1269", "languages": [], "modality": null, "name": "ConvBench", "openness": "open", "publisher": "Shanghai AI Laboratory", "released": "2024-03-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1269-convbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ConvBench", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "COPA is a causal inference task, which requires to select the correct causal relation based on the given premise. COPA是一个因果推断任务，要求根据给定的前提，选择正确的因果关系。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": "2019-05-02", "first_score_source_reference": {"basis": "score_publication", "date_precision": "score_publication", "note": "The original benchmark release day is absent from the crawl. Table 2 provides the earliest dated numeric language-model evaluation retained in this date archive for the source-linked COPA task; it is not a benchmark release date.", "reported_at": "2019-05-02", "score_evidence": {"locator": "Table 2, BERT row, COPA column", "metric": "accuracy", "model": "BERT-large-cased (fine-tuned)", "unit": "percent", "value": 69}, "source_key": "opencompass:529", "source_url": "https://arxiv.org/abs/1905.00537v1"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:529", "languages": [], "modality": null, "name": "COPA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-529-copa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/COPA", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "RAG", "大语言模型", "LLM", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Researchers present a large-scale conversational RAG benchmark named CORAL and propose a unified framework for standardizing and evaluating various conversational RAG baselines. CORAL 是一个大规模对话 RAG 基准，包含一个统一框架，用于标准化和评估各种对话 RAG 基线。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1622", "languages": [], "modality": null, "name": "CORAL", "openness": "open", "publisher": "RUC, BAAI, etc.", "released": "2024-10-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1622-coral", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CORAL", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CorpusQA is a multi-document, free-form long-context question answering benchmark in which a model must retrieve and reason over information distributed across a large corpus to produce open-ended answers that are scored by an LLM judge.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:corpusqa:mai-thinking-1", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:corpusqa", "languages": [], "modality": "text", "name": "CorpusQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.82, "raw_min": 0.82, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:corpusqa:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500"}, "unit": null}, "slug": "llm-stats-corpusqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CorpusQA 1M is a long-context question answering benchmark designed to evaluate models at approximately 1 million token contexts. Models are scored on accuracy when retrieving and reasoning over information distributed across an extremely long input corpus.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:corpusqa-1m:deepseek-v4-flash-0423", "reported_at": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:corpusqa-1m", "languages": [], "modality": "text", "name": "CorpusQA 1M", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.62, "raw_min": 0.593, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:corpusqa-1m:deepseek-v4-pro-max", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500"}, "unit": null}, "slug": "llm-stats-corpusqa-1m", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CountBench evaluates object counting capabilities in visual understanding.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:countbench:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:countbench", "languages": [], "modality": "image", "name": "CountBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.978, "display_multiplier": 1, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.978, "raw_min": 0.725, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:countbench:qwen3.5-27b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500"}, "unit": null}, "slug": "llm-stats-countbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CountQA is a benchmark for visual object counting and quantity reasoning over images.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:countqa:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:countqa", "languages": [], "modality": "multimodal", "name": "CountQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.77, "raw_min": 0.77, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:countqa:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500"}, "unit": null}, "slug": "llm-stats-countqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["speech_to_text", "language", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CoVoST 2 is a large-scale multilingual speech translation corpus derived from Common Voice, covering translations from 21 languages into English and from English into 15 languages. The dataset contains 2,880 hours of speech with 78K speakers for speech translation research.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:covost2:gemini-2.0-flash", "reported_at": "2024-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:covost2", "languages": [], "modality": "audio", "name": "CoVoST2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 40.699999999999996, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.407, "raw_min": 0.384, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:covost2:nova-2-omni", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500"}, "unit": null}, "slug": "llm-stats-covost2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500", "unit": null}, {"aliases": [], "categories": ["speech_to_text", "language", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CoVoST 2 English-to-Chinese subset is part of the large-scale multilingual speech translation corpus derived from Common Voice. This subset focuses specifically on English to Chinese speech translation tasks within the broader CoVoST 2 dataset.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:covost2-en-zh:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:covost2-en-zh", "languages": [], "modality": "audio", "name": "CoVoST2 en-zh", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.414, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.414, "raw_min": 0.414, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:covost2-en-zh:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500"}, "unit": null}, "slug": "llm-stats-covost2-en-zh", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500", "unit": null}, {"aliases": [], "categories": ["productivity", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CoWorkBench is Qwen's internal cowork benchmark for evaluating long-horizon office and productivity agent tasks across domains such as computer science, finance, law, and medicine.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:coworkbench:qwen3.7-max", "reported_at": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:coworkbench", "languages": [], "modality": "text", "name": "CoWorkBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.8, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.748, "raw_min": 0.651, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:coworkbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500"}, "unit": null}, "slug": "llm-stats-coworkbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "Problem Similarity", "Competitive Programming", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Programming contests are long used to evaluate algorithmic thinking and coding skills, and have recently become benchmarks for assessing large language models (LLMs). However, the rapid expansion of problem sets has led to a surge in duplicate or highly similar problems, compromising fairness ... 编程竞赛长期用于评估算法与编程能力，近年来也被用于大语言模型的评测。但随着题库扩展，重复或相似题激增，影响竞赛公平性与模型评测效果。为此，本文提出“相似题目检索”任务，并构建统一检索基准数据集 CPRet，涵盖题目与代码的四类检索任务，包含自动爬取和人工标注的数据样本。同时，设计并训练了两种检索模型 CPRetriever-Code 与 CPRetriever-Prob，显著提升检索效果。实验还发现相似题会提高模型得分、减小模型差异，强调了评测中引入“相似性感知”的必要性。我们还发布了开源检索平台，支持重复题检测与相似题推荐。项目地址：https://github.com/coldchair/", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1972", "languages": [], "modality": null, "name": "CPRet", "openness": "unknown", "publisher": "Tsinghua University, Shanghai AI Laboratory", "released": "2025-05-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1972-cpret", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CPRet", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "finance", "economics"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CRAG (Comprehensive RAG Benchmark) is a factual question answering benchmark consisting of 4,409 question-answer pairs across 5 domains (finance, sports, music, movie, open domain) and 8 question categories. The benchmark includes mock APIs to simulate web and Knowledge Graph search, designed to represent the diverse and dynamic nature of real-world QA tasks with temporal dynamism ranging from years to seconds. It evaluates retrieval-augmented generation systems for trustworthy question answering.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:crag:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:crag", "languages": [], "modality": "text", "name": "CRAG", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 50.3, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.503, "raw_min": 0.431, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:crag:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500"}, "unit": null}, "slug": "llm-stats-crag", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "RAG", "大语言模型", "LLM", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The Comprehensive RAG Benchmark (CRAG) is a rich and comprehensive factual question answering benchmark designed to advance research in RAG. Besides question-answer pairs, CRAG provides mock APIs to simulate web and knowledge graph search. CRAG是一个丰富且全面的基于事实的问题回答基准，旨在推进 RAG 研究。除了问答对之外，CRAG 还提供了模拟网页和知识图谱搜索的模拟 API。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1626", "languages": [], "modality": null, "name": "CRAG", "openness": "restricted", "publisher": "Meta Reality Labs,  FAIR, Meta, etc.", "released": "2024-06-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1626-crag", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CRAG", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "长文本", "Long-Context", "创作", "Creation", "多模态模型", "VLM", "长上下文", "Long Context", "语言生成", "Generation", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A multimodal benchmark specifically designed to evaluate the creative capabilities of MLLMs. It features three main aspects: 1. Comprehensive Creation Benchmark for MLLM and LLM. 2. Robust Evaluation Methodology. 3. Attractive Experiment Insight. 专为评估 多模态大模型 的创作能力而设计的多模态基准。采用两个不同指标对模型的基础感知能力和深层次视觉创作能力进行评估，采用GPT-4o作为评判模型进行评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1643", "languages": [], "modality": "multimodal", "name": "Creation-MMBench", "openness": "unknown", "publisher": "opencompass", "released": "2025-03-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1643-creation-mmbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Creation-MMBench", "unit": null}, {"aliases": [], "categories": ["creativity", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "EQ-Bench Creative Writing v3 is an LLM-judged creative writing benchmark that evaluates models across 32 writing prompts with 3 iterations per prompt. Uses a hybrid scoring system combining rubric assessment and Elo ratings through pairwise comparisons. Challenges models in areas like humor, romance, spatial awareness, and unique perspectives to assess emotional intelligence and creative writing capabilities.", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:creative-writing-v3:qwen3-235b-a22b-instruct-2507", "reported_at": "2025-07-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:creative-writing-v3", "languages": [], "modality": "text", "name": "Creative Writing v3", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.5, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.875, "raw_min": 0.761, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:creative-writing-v3:qwen3-235b-a22b-instruct-2507", "reported_date": "2025-07-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500"}, "unit": null}, "slug": "llm-stats-creative-writing-v3", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CreativeWork evaluates agents on open-ended creative production tasks within realistic tool and application environments, measuring the quality and completeness of generated deliverables.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:creativework:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:creativework", "languages": [], "modality": "multimodal", "name": "CreativeWork", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 42.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.425, "raw_min": 0.345, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:creativework:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500"}, "unit": null}, "slug": "llm-stats-creativework", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CREW-WILDFIRE is a benchmark designed to evaluate large language model-based multi-agent systems’ collaboration capabilities in complex, dynamic tasks, targeting agentic frameworks with perception, planning, and execution abilities. CREW-WILDFIRE 是一个用于评估基于大语言模型的多智能体系统在复杂动态任务中协作能力的基准，面向具备感知、规划与执行能力的智能体框架。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2046", "languages": [], "modality": null, "name": "CREW-WILDFIRE", "openness": "unknown", "publisher": "Duke University , Army Research Laboratory", "released": "2025-07-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2046-crew-wildfire", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CREW-WILDFIRE", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CriticBench, a novel benchmark designed to comprehensively and reliably evaluate four key critique ability dimensions of LLMs: feedback, comparison, refinement and meta-feedback. CriticBench encompasses nine diverse tasks, each assessing the LLMs' ability to critique responses at varying levels of quality granularity. CriticBench是一个新颖的基准，旨在全面可靠地评估LLM的四个关键批判能力维度。CriticBench包括九项不同的任务，每项任务都评估LLM在不同质量粒度水平上对响应进行批评的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:568", "languages": [], "modality": null, "name": "CriticBench", "openness": "unknown", "publisher": null, "released": "2024-02-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-568-criticbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CriticBench", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "reasoning", "science"], "collected_at": "2026-08-25T10:41:06Z", "description": "Physics reasoning", "evidence_summary": {"document_count": 1, "model_count": 492, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:critpt:217b34ec-5920-4fc1-8886-6a70a324837d", "reported_at": "2023-09-27", "source_url": "https://artificialanalysis.ai/evaluations/critpt"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:critpt", "languages": [], "modality": null, "name": "CritPt", "openness": "unknown", "publisher": null, "released": "2025-11-21", "released_reference": {"basis": "release_announcement", "note": "The project announces its public challenge dataset and evaluation pipeline, and its debut on Artificial Analysis, on November 21. Earlier scored model releases do not date this benchmark.", "source_key": "artificial-analysis:critpt", "source_url": "https://github.com/CritPt-Benchmark/CritPt"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 492, "score_direction": "higher_is_better", "score_summary": {"display_max": 32.2857142857143, "display_multiplier": 100, "model_count": 492, "model_count_basis": "source_model_id", "numeric_count": 492, "raw_max": 0.322857142857143, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:critpt:d93edfe8-bf35-49ad-b56e-b18116142a1c", "reported_date": "2026-07-09", "source_url": "https://artificialanalysis.ai/evaluations/critpt"}, "unit": null}, "slug": "artificial-analysis-critpt", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/critpt", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CritPT is a challenging reasoning benchmark reported by Qwen for evaluating frontier mathematical and critical problem-solving capability.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:critpt:qwen3.7-max", "reported_at": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:critpt", "languages": [], "modality": "text", "name": "CritPT", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 16.7, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.167, "raw_min": 0.031, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:critpt:glm-5.2", "reported_date": "2026-06-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500"}, "unit": null}, "slug": "llm-stats-critpt", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500", "unit": null}, {"aliases": ["CritPt"], "categories": ["science"], "collected_at": null, "description": "Research-level physics reasoning; small expert-authored set.", "evidence_summary": {"document_count": 3, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000critpt\u0000moonshot_kimi_k3_model_card\u0000critpt\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "first_score_reported_at": "2026-06-13", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000critpt\u0000moonshot_kimi_k3_model_card\u0000critpt\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:critpt", "languages": [], "modality": null, "name": "CritPt", "openness": "unknown", "publisher": null, "released": "2025-09-30", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:critpt", "source_url": "https://arxiv.org/abs/2509.26574"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 23.4, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 23.4, "raw_min": 16.9, "source_reference": {"obs_id": "curated\u0000critpt\u0000moonshot_kimi_k3_model_card\u0000critpt\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000critpt\u0000moonshot_kimi_k3_model_card\u0000critpt\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "critpt", "source": "model_reports", "source_url": "https://arxiv.org/abs/2509.26574", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CrossVid evaluates cross-video reasoning, requiring models to integrate information across multiple videos.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:crossvid:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:crossvid", "languages": [], "modality": "multimodal", "name": "CrossVid", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.65, "raw_min": 0.632, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:crossvid:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500"}, "unit": null}, "slug": "llm-stats-crossvid", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CrossWordBench, a benchmark designed to evaluate the reasoning capabilities of both LLMs and LVLMs through the medium of crossword puzzles. CrossWordBench，这是一个基准测试，旨在通过填字游戏的方式来评估LLMs和LVLMs的推理能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1739", "languages": [], "modality": "multimodal", "name": "CrossWordBench", "openness": "restricted", "publisher": "CMU, WUSTL,UW,UIUC", "released": "2025-03-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1739-crosswordbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CrossWordBench", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CrowS-Pairs has 1508 examples that cover stereotypes dealing with nine types of bias, like race, religion, and age. In CrowS-Pairs a model is presented with two sentences: one that is more stereotyping and another that is less stereotyping. CrowS-Pairs 包含 1508 个示例，涵盖与九种偏见类型相关的刻板印象，例如种族、宗教和年龄。在 CrowS-Pairs 中，模型会接收到两句话：一句是更具刻板印象的，另一句则是较少刻板印象的。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1123", "languages": [], "modality": null, "name": "Crows-Pairs", "openness": "unknown", "publisher": "New York University", "released": "2020-09-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1123-crows-pairs", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Crows-Pairs", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CRPE is a benchmark designed to quantitatively evaluate the object recognition and relation comprehension ability of models. It consists of four splits, and the evaluation is formulated as single-choice questions. CRPE用于定量评估多模态大模型的对象识别和关系理解能力，分为四部分，以单选题形式呈现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1500", "languages": [], "modality": "multimodal", "name": "CRPE", "openness": "open", "publisher": "Shanghai AI Laboratory", "released": "2024-02-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1500-crpe", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CRPE", "unit": null}, {"aliases": [], "categories": ["reasoning", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Clinical reasoning problems evaluation benchmark for assessing diagnostic reasoning and medical knowledge application capabilities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:crperelation:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:crperelation", "languages": [], "modality": "text", "name": "CRPErelation", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.765, "raw_min": 0.765, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:crperelation:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500"}, "unit": null}, "slug": "llm-stats-crperelation", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CRUXEval-O (output prediction) is part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate AI models' capabilities in code reasoning, understanding, and execution. The benchmark tests models' ability to predict correct function outputs given function code and inputs, focusing on short problems that a good human programmer should be able to solve in a minute.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:crux-o:qwen3-235b-a22b", "reported_at": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:crux-o", "languages": [], "modality": "text", "name": "CRUX-O", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.79, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.79, "raw_min": 0.79, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:crux-o:qwen3-235b-a22b", "reported_date": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500"}, "unit": null}, "slug": "llm-stats-crux-o", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CRUXEval input prediction task with Chain of Thought (CoT) prompting. Part of the CRUXEval benchmark for code reasoning, understanding, and execution evaluation. Given a Python function and its expected output, the task is to predict the appropriate input using chain-of-thought reasoning. Consists of 800 Python functions (3-13 lines) designed to evaluate code comprehension and reasoning capabilities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cruxeval-input-cot:qwen-2.5-coder-7b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cruxeval-input-cot", "languages": [], "modality": "text", "name": "CRUXEval-Input-CoT", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.49999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.565, "raw_min": 0.565, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cruxeval-input-cot:qwen-2.5-coder-7b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500"}, "unit": null}, "slug": "llm-stats-cruxeval-input-cot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CruxEval-O is the output prediction task of the CRUXEval benchmark, designed to evaluate code reasoning, understanding, and execution capabilities. It consists of 800 Python functions (3-13 lines) where models must predict the output given a function and input. The benchmark tests fundamental code execution reasoning abilities and goes beyond simple code generation to assess deeper understanding of program behavior.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cruxeval-o:codestral-22b", "reported_at": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cruxeval-o", "languages": [], "modality": "text", "name": "CruxEval-O", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.300000000000004, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.513, "raw_min": 0.513, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cruxeval-o:codestral-22b", "reported_date": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500"}, "unit": null}, "slug": "llm-stats-cruxeval-o", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CRUXEval-O (output prediction) with Chain-of-Thought prompting. Part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate code reasoning, understanding, and execution capabilities. The output prediction task requires models to predict the output of a given Python function with specific inputs, evaluated using chain-of-thought reasoning methodology.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cruxeval-output-cot:qwen-2.5-coder-7b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cruxeval-output-cot", "languages": [], "modality": "text", "name": "CRUXEval-Output-CoT", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.00000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.56, "raw_min": 0.56, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cruxeval-output-cot:qwen-2.5-coder-7b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500"}, "unit": null}, "slug": "llm-stats-cruxeval-output-cot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "代码", "Code", "数学", "Math", "大语言模型", "LLM", "逻辑推理", "代码工程", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CS-Bench, the first bilingual (Chinese-English) benchmark dedicated to evaluating the performance of LLMs in computer science. CS-Bench comprises approximately 5K meticulously curated test samples, covering 26 subfields across 4 key areas of computer science, encompassing various task forms and divisions of knowledge and reasoning. CS-Bench 第一个专门用于评估LLMs在计算机科学中表现的双语（中英文）基准。CS-Bench包括约5,000个精心策划的测试样本，涵盖了计算机科学4个关键领域中的26个子领域，并包括各种任务形式和知识推理的划分。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:924", "languages": ["English", "Chinese"], "modality": null, "name": "CS-Bench", "openness": "restricted", "publisher": null, "released": "2024-06-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-924-cs-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CS-Bench", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "知识", "Knowledge", "安全", "Safety", "大语言模型", "LLM", "知识储备", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CS-Eval is a large language model cybersecurity capability evaluation suite jointly established by Alibaba Security, Fudan University, and the University of Chinese Academy of Sciences. The dataset encompasses 11 major cybersecurity categories and 42 subcategories, offering comprehensive assessment CS-Eval 是由阿里安全、复旦大学和中国科学院大学联合建立的大模型网络安全能力评测集。数据集覆盖11个网络安全大类领域、42个子类领域，提供知识型和实战型的综合评估任务，支持用户自主评测，同时为大模型落地网络安全提供参考和启发。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1219", "languages": ["English", "Chinese"], "modality": null, "name": "CS-Eval", "openness": "restricted", "publisher": "阿里巴巴集团安全部，复旦大学，中国科学院大学", "released": "2024-05-31", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1219-cs-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CS-Eval", "unit": null}, {"aliases": [], "categories": ["language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Chinese SimpleQA is the first comprehensive Chinese benchmark to evaluate the factuality ability of language models to answer short questions. It contains 3,000 high-quality questions spanning 6 major topics with 99 diverse subtopics, designed to assess Chinese factual knowledge across humanities, science, engineering, culture, and society.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:csimpleqa:deepseek-v3", "reported_at": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:csimpleqa", "languages": [], "modality": "text", "name": "CSimpleQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.39999999999999, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.844, "raw_min": 0.648, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:csimpleqa:deepseek-v4-pro-max", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500"}, "unit": null}, "slug": "llm-stats-csimpleqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CSL is a large-scale Chinese Scientific Literature dataset, which contains the titles, abstracts, keywords and academic fields of 396k papers. CSL是一个大规模的中文科技文献数据集，包含 39.6 万篇论文的标题、摘要、关键词和学术领域信息。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:519", "languages": ["Chinese"], "modality": null, "name": "CSL", "openness": "unknown", "publisher": null, "released": "2022-09-12", "released_reference": {"basis": "paper_first_version", "note": "The source describes the 396k-paper scientific literature dataset introduced here.", "source_key": "opencompass:519", "source_url": "https://arxiv.org/abs/2209.05034"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-519-csl", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CSL", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CSTS is a synthetic benchmark for evaluating correlation structure discovery in multivariate time series. It features 23 distinct correlation structures with systematic data variations (distribution shifts, sparsification, downsampling) and provides ground truth labels for validation. CSTS is a synthetic benchmark for evaluating correlation structure discovery in multivariate time series. It features 23 distinct correlation structures with systematic data variations (distribution shifts, sparsification, downsampling) and provides ground truth labels for validation.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1833", "languages": [], "modality": null, "name": "CSTS", "openness": "open", "publisher": "University of Bristol,University of Nanjing,etc", "released": "2025-05-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1833-csts", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CSTS", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "NeurIPS 2024", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CTIBench is a benchmark designed to assess LLMs' performance in CTI (Cyber threat intelligence) applications. It includes multiple datasets focused on evaluating knowledge acquired by LLMs in the cyber-threat landscape. CTIBench旨在评估LLM在CTI（网络安全情报）场景下的能力，包含多个数据集，专注于评测LLM在网络威胁环境中获得的知识。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1268", "languages": [], "modality": null, "name": "CTIBench", "openness": "restricted", "publisher": "Rochester Institute of Technology", "released": "2024-06-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1268-ctibench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CTIBench", "unit": null}, {"aliases": ["CursorBench", "Cursor Bench", "CursorBench 3.2"], "categories": ["coding_agent"], "collected_at": null, "description": "Maintained by Cursor and measured inside the Cursor product, so it reflects that scaffold rather than the bare model.", "evidence_summary": {"document_count": 2, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:cursor_bench", "languages": [], "modality": null, "name": "CursorBench", "openness": "unknown", "publisher": null, "released": "2025-10-29", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:cursor_bench", "source_url": "https://cursor.com/blog/cursor-bench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "cursor_bench", "source": "model_reports", "source_url": "https://cursor.com/blog/cursor-bench", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CursorBench v3.2 evaluates coding agents on interactive software engineering tasks in the Cursor environment.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cursorbench-3.2:grok-4.6", "reported_at": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cursorbench-3.2", "languages": [], "modality": "text", "name": "CursorBench v3.2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 69.89999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.699, "raw_min": 0.699, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cursorbench-3.2:grok-4.6", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500"}, "unit": null}, "slug": "llm-stats-cursorbench-3-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CVDP is a next-generation benchmark for evaluating large language models (LLMs) and agents in hardware design and verification, comprising 783 problems across 13 task categories, including RTL generation, verification, debugging, specification alignment, and technical Q&A. CVDP 是一个面向大型语言模型（LLM）和智能体的下一代硬件设计与验证评测基准，涵盖 13 类任务共 783 个问题，涉及 RTL 生成、验证、调试、规范对齐和技术问答等。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1985", "languages": [], "modality": null, "name": "CVDP", "openness": "unknown", "publisher": "NVIDIA", "released": "2025-06-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1985-cvdp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CVDP", "unit": null}, {"aliases": ["CVE-Bench", "CVEBench"], "categories": ["security"], "collected_at": null, "description": "Real CVE exploitation in sandboxes. Vendors report it as a safety ceiling rather than a capability to maximize.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:cvebench", "languages": [], "modality": null, "name": "CVE-Bench", "openness": "unknown", "publisher": null, "released": "2025-03-21", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:cvebench", "source_url": "https://github.com/uiuc-kang-lab/cve-bench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "cvebench", "source": "model_reports", "source_url": "https://github.com/uiuc-kang-lab/cve-bench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "知识", "Knowledge", "NeurIPS 2024", "多模态模型", "VLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CVQA is a new culturally-diverse multilingual Visual Question Answering benchmark, designed to cover a rich set of languages and cultures. It includes culturally-driven images and 10k questions from across 30 countries on 4 continents. CVQA是一种新的文化多元化多语言视觉问答基准，旨在涵盖丰富的语言和文化，包括来自四大洲30个国家/地区的文化向图像和10000个问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1239", "languages": ["English", "Japanese", "Multilingual"], "modality": "multimodal", "name": "CVQA", "openness": "restricted", "publisher": "MBZUAI", "released": "2024-06-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1239-cvqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CVQA", "unit": null}, {"aliases": [], "categories": ["image-generation", "language", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CVTG-2K (Chinese Visual Text Generation 2K) is a benchmark for evaluating text-to-image models on their ability to accurately render text within generated images. It measures Word Accuracy, Normalized Edit Distance (NED), and CLIPScore across 2,000 prompts.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cvtg-2k", "languages": [], "modality": "image", "name": "CVTG-2K", "openness": "unknown", "publisher": null, "released": "2025-03-30", "released_reference": {"basis": "paper_first_version", "note": "TextCrafter introduces the complex visual text generation benchmark. The source's Chinese-only name expansion is not used to identify a separate instrument.", "source_key": "llm-stats:cvtg-2k", "source_url": "https://arxiv.org/abs/2503.23461"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-cvtg-2k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cvtg-2k?top_n=500", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CyBench is a suite of Capture-the-Flag (CTF) challenges measuring agentic cyber attack capabilities. It evaluates dual-use cybersecurity knowledge and measures the 'unguided success rate', where agents complete tasks end-to-end without guidance on appropriate subtasks.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cybench:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cybench", "languages": [], "modality": "text", "name": "CyBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 100.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 1.0, "raw_min": 0.39, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cybench:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500"}, "unit": null}, "slug": "llm-stats-cybench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CyberGym is a benchmark for evaluating AI agents on cybersecurity tasks, testing their ability to identify vulnerabilities, perform security analysis, and complete security-related challenges in a controlled environment.", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cybergym:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cybergym", "languages": [], "modality": "text", "name": "CyberGym", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.5, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.845, "raw_min": 0.413, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cybergym:glm-5.3", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500"}, "unit": null}, "slug": "llm-stats-cybergym", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500", "unit": null}, {"aliases": [], "categories": ["security"], "collected_at": null, "description": "Real-world cybersecurity tasks; tool permissions and time limits are part of the result, not incidental to it.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000cybergym\u0000tencent_hy4_preview\u0000cybergym\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000cybergym\u0000tencent_hy4_preview\u0000cybergym\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:cybergym", "languages": [], "modality": null, "name": "CyberGym", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 78.4, "raw_min": 78.4, "source_reference": {"obs_id": "curated\u0000cybergym\u0000tencent_hy4_preview\u0000cybergym\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000cybergym\u0000tencent_hy4_preview\u0000cybergym\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "cybergym", "source": "model_reports", "source_url": "https://openreview.net/forum?id=2YvbLQEdYt", "unit": "percent"}, {"aliases": [], "categories": ["安全", "Safety", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CyberSecEval is a comprehensive benchmark developed to help bolster the cybersecurity of LLMs. It provides a thorough evaluation in two crucial security domains: the propensity to generate insecure code and the level of compliance when asked to assist in cyberattacks. CyberSecEval旨在评估LLM的安全性，聚焦于大模型生成不安全代码的倾向以及当被要求协助网络攻击时的合规性水平。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1349", "languages": ["English", "Multilingual"], "modality": null, "name": "CyberSecEval", "openness": "unknown", "publisher": "Meta", "released": "2023-12-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1349-cyberseceval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CyberSecEval", "unit": null}, {"aliases": [], "categories": ["safety", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "CyberSecEval 4 is an evaluation suite covering cybersecurity-related capabilities and risks of large language models. The insecure-code-generation tracks measure whether a model produces vulnerable code: the Instruct track presents coding requests designed to elicit known insecure patterns, while the Autocomplete track prompts the model with code context leading up to a known insecure pattern, with vulnerabilities detected via static analysis.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cyberseceval-4:mai-thinking-1", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cyberseceval-4", "languages": [], "modality": "text", "name": "CyberSecEval 4", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.63, "raw_min": 0.63, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cyberseceval-4:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500"}, "unit": null}, "slug": "llm-stats-cyberseceval-4", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500", "unit": null}, {"aliases": [], "categories": ["safety"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Cybersecurity Capture the Flag (CTF) benchmark for evaluating LLMs in offensive security challenges. Contains diverse cybersecurity tasks including cryptography, web exploitation, binary analysis, and forensics to assess AI capabilities in cybersecurity problem-solving.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:cybersecurity-ctfs:o1-mini", "reported_at": "2024-09-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:cybersecurity-ctfs", "languages": [], "modality": "text", "name": "Cybersecurity CTFs", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.60000000000001, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.776, "raw_min": 0.287, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:cybersecurity-ctfs:gpt-5.3-codex", "reported_date": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500"}, "unit": null}, "slug": "llm-stats-cybersecurity-ctfs", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DABstep is a benchmark for evaluating AI agents’ multi-step reasoning and planning abilities in realistic data analysis tasks, targeting code-executing language model agents. DABstep 是一个用于评估 AI 智能体在现实多步骤数据分析任务中推理与规划能力的基准，面向具备代码执行能力的语言模型代理。该基准包含450多个任务，源自金融分析平台，涵盖结构化数据处理、非结构化文档理解和跨源信息整合。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:2030", "languages": [], "modality": null, "name": "DABstep", "openness": "unknown", "publisher": "Adyen , HuggingFace", "released": "2025-06-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2030-dabstep", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DABstep", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DailyOmni evaluates multimodal models on daily-life video understanding tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:dailyomni:mimo-v2.5", "reported_at": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:dailyomni", "languages": [], "modality": "multimodal", "name": "DailyOmni", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.835, "raw_min": 0.835, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:dailyomni:mimo-v2.5", "reported_date": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500"}, "unit": null}, "slug": "llm-stats-dailyomni", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "ACL 2024", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DebugBench is an LLM debugging benchmark consisting of 4,253 instances. It covers four major bug categories and 18 minor types in C++, Java, and Python. DebugBench 是一个包含 4,253 个实例的 LLM 调试基准。它涵盖了 C++、Java 和 Python 中的四个主要错误类别和 18 种次要类型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1078", "languages": [], "modality": null, "name": "DebugBench", "openness": "open", "publisher": "THUNLP", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1078-debugbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DebugBench", "unit": null}, {"aliases": [], "categories": ["productivity", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DECK-Bench is Moonshot AI's internal evaluation of agents on creating and reasoning over presentation-style knowledge-work artifacts.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:deck-bench:kimi-k3", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:deck-bench", "languages": [], "modality": "text", "name": "DECK-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.735, "raw_min": 0.735, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:deck-bench:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-deck-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "多模态模型", "VLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Deepfake-Eval-2024 is an in-the-wild deepfake dataset. Deepfake-Eval-2024 contains 44 hours of videos, 56.5 hours of audio, and 1,975 images, encompassing contemporary manipulation technologies, diverse media content, 88 different website sources, and 52 different languages. Deepfake-Eval-2024 是一个真实场景的深度伪造数据集。该数据集包含44小时的视频、56.5小时的音频和1,975张图像，涵盖当代篡改技术、多样化的媒体内容、来自88个不同网站来源的素材以及52种不同语言。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1605", "languages": [], "modality": null, "name": "Deepfake-Eval-2024", "openness": "unknown", "publisher": "TrueMedia.org, etc.", "released": "2025-03-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1605-deepfake-eval-2024", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Deepfake-Eval-2024", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DeepPlanning evaluates LLMs on complex multi-step planning tasks requiring long-horizon reasoning, goal decomposition, and strategic decision-making.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:deep-planning:qwen3.5-397b-a17b", "reported_at": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:deep-planning", "languages": [], "modality": "text", "name": "DeepPlanning", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.3, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.623, "raw_min": 0.176, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:deep-planning:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500"}, "unit": null}, "slug": "llm-stats-deep-planning", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DeepResearch Bench is a comprehensive benchmark designed to evaluate large language model (LLM) agents on complex research tasks. DeepResearch Bench 是一个面向大型语言模型（LLM）智能体的综合性评测基准，专为评估其在复杂研究任务中的表现而设计。 该基准包含 100 个由 22 个领域的专家精心设计的博士级研究任务，涵盖多领域的深度研究需求。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1987", "languages": ["English", "Chinese"], "modality": null, "name": "DeepResearchBench", "openness": "unknown", "publisher": "University of Science and Technology of China , Metastone Technology, Beijing", "released": "2025-06-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1987-deepresearchbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DeepResearchBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DeepSearchQA is a benchmark for evaluating deep search and question-answering capabilities, testing models' ability to perform multi-hop reasoning and information retrieval across complex knowledge domains.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:deepsearchqa:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:deepsearchqa", "languages": [], "modality": "text", "name": "DeepSearchQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.0, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.95, "raw_min": 0.746, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:deepsearchqa:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500"}, "unit": null}, "slug": "llm-stats-deepsearchqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500", "unit": null}, {"aliases": ["DeepSearchQA"], "categories": ["agent"], "collected_at": null, "description": "Live-web deep research; F1 grading depends on the reference answer set.", "evidence_summary": {"document_count": 2, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000deepsearchqa\u0000moonshot_kimi_k3_model_card\u0000deepsearchqa\u0000F1, reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "first_score_reported_at": "2026-06-13", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000deepsearchqa\u0000moonshot_kimi_k3_model_card\u0000deepsearchqa\u0000F1, reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:deepsearchqa", "languages": [], "modality": null, "name": "DeepSearchQA", "openness": "unknown", "publisher": null, "released": "2026-02-10", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:deepsearchqa", "source_url": "https://huggingface.co/datasets/PokeeAI/DeepSearchQA"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.0, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 95.0, "raw_min": 95.0, "source_reference": {"obs_id": "curated\u0000deepsearchqa\u0000moonshot_kimi_k3_model_card\u0000deepsearchqa\u0000F1, reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000deepsearchqa\u0000moonshot_kimi_k3_model_card\u0000deepsearchqa\u0000F1, reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "deepsearchqa", "source": "model_reports", "source_url": "https://huggingface.co/datasets/PokeeAI/DeepSearchQA", "unit": "percent"}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DeepSWE is a software engineering agent benchmark evaluated with the mini-swe-agent harness, where each task is solved in an isolated container with no internet access. It measures an agent's ability to autonomously resolve real-world coding issues end to end.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:deepswe:glm-5.2", "reported_at": "2026-06-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:deepswe", "languages": [], "modality": "text", "name": "DeepSWE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.7, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.727, "raw_min": 0.23, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:deepswe:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500"}, "unit": null}, "slug": "llm-stats-deepswe", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500", "unit": null}, {"aliases": ["DeepSWE"], "categories": ["coding_agent"], "collected_at": null, "description": "Mini-swe-agent harness, temperature=0.95, top_p=1.0, max_new_tokens=64k under 1M context.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000deepswe\u0000zai_glm_5_3_flash_model_card\u0000deepswe_v1.1\u0000DeepSWE v1.1, mini-swe-agent harness, temp 0.95, top_p 1.0, timeout 6h, 400K context\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000deepswe\u0000zai_glm_5_3_flash_model_card\u0000deepswe_v1.1\u0000DeepSWE v1.1, mini-swe-agent harness, temp 0.95, top_p 1.0, timeout 6h, 400K context\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:deepswe", "languages": [], "modality": null, "name": "DeepSWE", "openness": "unknown", "publisher": null, "released": "2025-09-15", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:deepswe", "source_url": "https://github.com/deepswe/deepswe"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.3, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 64.3, "raw_min": 63.4, "source_reference": {"obs_id": "curated\u0000deepswe\u0000tencent_hy4_preview\u0000deepswe\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000deepswe\u0000tencent_hy4_preview\u0000deepswe\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "deepswe", "source": "model_reports", "source_url": "https://github.com/deepswe/deepswe", "unit": "percent"}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DeepSWE 1.0 pass@1 benchmark as reported by Artificial Analysis using provider-specific harness runs.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:deepswe-1.0:grok-4.5", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:deepswe-1.0", "languages": [], "modality": "text", "name": "DeepSWE 1.0", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.62, "raw_min": 0.62, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:deepswe-1.0:grok-4.5", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500"}, "unit": null}, "slug": "llm-stats-deepswe-1-0", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DeepSWE 1.1 evaluates software engineering agents using the mini-swe-agent harness.", "evidence_summary": {"document_count": 1, "model_count": 25, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:deepswe-1.1:claude-sonnet-4-6", "reported_at": "2026-02-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:deepswe-1.1", "languages": [], "modality": "text", "name": "DeepSWE 1.1", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 25, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.0, "display_multiplier": 100, "model_count": 25, "model_count_basis": "source_model_id", "numeric_count": 25, "raw_max": 0.73, "raw_min": 0.12, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:deepswe-1.1:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500"}, "unit": null}, "slug": "llm-stats-deepswe-1-1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500", "unit": null}, {"aliases": [], "categories": ["healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Dermatology multiple choice question assessment benchmark for evaluating medical knowledge and diagnostic reasoning in dermatological conditions and treatments.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:dermmcqa:medgemma-4b-it", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:dermmcqa", "languages": [], "modality": "text", "name": "DermMCQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.718, "raw_min": 0.718, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:dermmcqa:medgemma-4b-it", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500"}, "unit": null}, "slug": "llm-stats-dermmcqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Design2Code evaluates the ability to generate code (HTML/CSS/JS) from visual designs.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:design2code:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:design2code", "languages": [], "modality": "image", "name": "Design2Code", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.948, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.948, "raw_min": 0.934, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:design2code:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500"}, "unit": null}, "slug": "llm-stats-design2code", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DICE-BENCH is a benchmark for evaluating large language models’ tool-use capabilities in multi-round, multi-party dialogues, focusing on function selection, parameter filling, and dialogue context comprehension. DICE-BENCH 是一个评估大型语言模型在多轮、多方对话中工具调用能力的基准，涵盖函数选择、参数填充和对话上下文理解等维度。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2032", "languages": [], "modality": null, "name": "DICE-BENCH", "openness": "open", "publisher": "IPAI,Seoul National University ,etc", "released": "2025-06-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2032-dice-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DICE-BENCH", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VLB provides a robust and comprehensive assessment for LVLMs with reduced data contamination and flexible complexity. Based on LlavaBench and MMvet, we have curated two more challenging versions of the datasets: LlavaBench_hard and MMvet_hard. These are the hardest multimodal combinations. VLB 为 LVLMs 提供了一种稳健且全面的评估，降低了数据污染并具有灵活的复杂性。基于 LlavaBench 和 MMvet，我们精心制作了两个更具挑战性的数据集版本：LlavaBench_hard 和 MMvet_hard。这些是我们动态策略中最具挑战性的多模态组合（V1+L4）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1648", "languages": [], "modality": "multimodal", "name": "DME", "openness": "unknown", "publisher": "SJTU, Shanghai AI Laboratory, etc.", "released": "2024-10-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1648-dme", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DME", "unit": null}, {"aliases": [], "categories": ["multimodal", "image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A dataset for Visual Question Answering on document images containing 50,000 questions defined on 12,000+ document images. The benchmark tests AI's ability to understand document structure and content, requiring models to comprehend document layout and perform information retrieval to answer questions about document images.", "evidence_summary": {"document_count": 1, "model_count": 28, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:docvqa:grok-1.5", "reported_at": "2024-03-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:docvqa", "languages": [], "modality": "multimodal", "name": "DocVQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 28, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.39999999999999, "display_multiplier": 100, "model_count": 28, "model_count_basis": "source_model_id", "numeric_count": 28, "raw_max": 0.964, "raw_min": 0.758, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:docvqa:qwen2.5-vl-72b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500"}, "unit": null}, "slug": "llm-stats-docvqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500", "unit": null}, {"aliases": ["DocVQA"], "categories": ["multimodal"], "collected_at": null, "description": "Saturated at the frontier; ANLS scoring is lenient to OCR near-misses.", "evidence_summary": {"document_count": 2, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:docvqa", "languages": [], "modality": null, "name": "DocVQA", "openness": "unknown", "publisher": null, "released": "2020-07-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:docvqa", "source_url": "https://www.docvqa.org/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "docvqa", "source": "model_reports", "source_url": "https://www.docvqa.org/", "unit": null}, {"aliases": [], "categories": ["multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DocVQA is a Visual Question Answering benchmark on document images containing 50,000 questions defined on 12,000+ document images. The benchmark focuses on understanding document structure and content to answer questions about various document types including letters, memos, notes, and reports from the UCSF Industry Documents Library.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:docvqatest:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:docvqatest", "languages": [], "modality": "multimodal", "name": "DocVQAtest", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.1, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.971, "raw_min": 0.942, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:docvqatest:qwen3-vl-235b-a22b-instruct", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500"}, "unit": null}, "slug": "llm-stats-docvqatest", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500", "unit": null}, {"aliases": [], "categories": ["instruction_following", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Doubao Multi-Turn Bench evaluates models on multi-turn conversational tasks, measuring context retention, instruction following, and coherent reasoning across extended dialogues.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:doubao-multi-turn-bench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:doubao-multi-turn-bench", "languages": [], "modality": "text", "name": "Doubao Multi-Turn Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 52.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.525, "raw_min": 0.49, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:doubao-multi-turn-bench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-doubao-multi-turn-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "LLM敏感性评估", "大语言模型", "LLM", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DOVE (Dataset Of Variation Evaluation) is a large-scale dataset containing prompt perturbations of various evaluation benchmarks. DOVE（变异评估数据集），这是一个大规模数据集，包含了各种评估基准的提示扰动。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1731", "languages": [], "modality": null, "name": "DOVE", "openness": "unknown", "publisher": "The Hebrew University of Jerusalem, IBM Research AI, MIT，etc.", "released": "2025-03-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1731-dove", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DOVE", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DRACO is a deep research benchmark that evaluates an agent's ability to gather, synthesize, and reason over information to answer complex research questions. Scores are based on official rubrics per question, with the final score being the average across all questions.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:draco:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:draco", "languages": [], "modality": "text", "name": "DRACO", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.22999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.7323, "raw_min": 0.7323, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:draco:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500"}, "unit": null}, "slug": "llm-stats-draco", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context"], "collected_at": null, "description": "Deep-research style long-context agent tasks; the retrieval and tool stack is part of the score.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000draco\u0000tencent_hy4_preview\u0000draco\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000draco\u0000tencent_hy4_preview\u0000draco\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:draco", "languages": [], "modality": null, "name": "DRACO", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.2, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 77.2, "raw_min": 77.2, "source_reference": {"obs_id": "curated\u0000draco\u0000tencent_hy4_preview\u0000draco\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000draco\u0000tencent_hy4_preview\u0000draco\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "draco", "source": "model_reports", "source_url": "https://huggingface.co/datasets/perplexity-ai/draco", "unit": "percent"}, {"aliases": [], "categories": ["知识", "Knowledge", "RAG", "大语言模型", "LLM", "知识储备", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DRAGON 是一个用于评估检索增强生成（RAG）系统在俄语新闻语境中事实性与检索能力的动态基准，支持对检索器与生成器组件的全面评估。 DRAGON is a dynamic benchmark for evaluating Retrieval-Augmented Generation (RAG) systems in Russian news contexts, supporting comprehensive assessment of both retriever and generator components.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2045", "languages": [], "modality": null, "name": "DRAGON", "openness": "unknown", "publisher": "SberAI , ITMO , MISIS , HSE , MWSAI", "released": "2025-07-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2045-dragon", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DRAGON", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DROP (Discrete Reasoning Over Paragraphs) is a reading comprehension benchmark requiring discrete reasoning over paragraph content. It contains crowdsourced, adversarially-created questions that require resolving references and performing discrete operations like addition, counting, or sorting, demanding comprehensive paragraph understanding beyond paraphrase-and-entity-typing shortcuts.", "evidence_summary": {"document_count": 1, "model_count": 30, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:drop:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:drop", "languages": [], "modality": "text", "name": "DROP", "openness": "unknown", "publisher": null, "released": "2019-03-01", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the DROP reading-comprehension benchmark.", "source_key": "opencompass:536", "source_url": "https://arxiv.org/abs/1903.00161"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 30, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.60000000000001, "display_multiplier": 100, "model_count": 30, "model_count_basis": "source_model_id", "numeric_count": 30, "raw_max": 0.916, "raw_min": 0.286, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:drop:deepseek-v3", "reported_date": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500"}, "unit": null}, "slug": "llm-stats-drop", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500", "unit": null}, {"aliases": ["DROP"], "categories": ["reasoning"], "collected_at": null, "description": "Legacy reading-comprehension baseline.", "evidence_summary": {"document_count": 6, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000drop\u0000google_gemini_1_5_report\u0000drop\u0000variable shots, F1\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "first_score_reported_at": "2024-03-08", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000drop\u0000google_gemini_1_5_report\u0000drop\u0000variable shots, F1\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:drop", "languages": [], "modality": null, "name": "DROP", "openness": "unknown", "publisher": null, "released": "2019-03-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:drop", "source_url": "https://allenai.org/data/drop"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.2, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 92.2, "raw_min": 74.9, "source_reference": {"obs_id": "curated\u0000drop\u0000deepseek_r1_report\u0000drop\u00003-shot F1\u0000DeepSeek-R1", "observation_id": "curated\u0000drop\u0000deepseek_r1_report\u0000drop\u00003-shot F1\u0000DeepSeek-R1", "reported_at": "2025-01-22", "reported_date": "2025-01-22", "source_id": "deepseek_r1_report", "source_url": "https://arxiv.org/abs/2501.12948"}, "unit": "percent"}, "slug": "drop", "source": "model_reports", "source_url": "https://allenai.org/data/drop", "unit": "percent"}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DROP is a QA dataset which tests comprehensive understanding of paragraphs. In this crowdsourced, adversarially-created, 96k question-answering benchmark, a system must resolve multiple references in a question, map them onto a paragraph, and perform discrete operations over them (such as addition, counting, or sorting). DROP 是一个测试段落综合理解能力的 QA 数据集。在这个众包、对抗性创建的 96K 问题解答基准中，系统必须解析问题中的多个引用，将它们映射到段落中，并对它们执行离散操作（如加法、计数或排序）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:536", "languages": [], "modality": null, "name": "DROP", "openness": "unknown", "publisher": null, "released": "2019-03-01", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the DROP reading-comprehension benchmark.", "source_key": "opencompass:536", "source_url": "https://arxiv.org/abs/1903.00161"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-536-drop", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DROP", "unit": null}, {"aliases": [], "categories": ["代码", "code", "大语言模型", "LLM", "代码工程", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DS-1000 is a code generation benchmark with a thousand data science questions spanning seven Python libraries that (1) reflects diverse, realistic, and practical use cases, (2) has a reliable metric, (3) defends against memorization by perturbing questions. DS-1000 是一个代码生成基准测试，包含一千个数据科学问题，涵盖七个Python库，其特点是（1）反映多样化、现实且实用的用例，（2）具有可靠的度量标准，（3）通过扰乱问题来防止记忆化。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:544", "languages": [], "modality": null, "name": "DS-1000", "openness": "unknown", "publisher": null, "released": "2022-11-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-544-ds-1000", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DS-1000", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Data Science Arena Code benchmark for evaluating LLMs on realistic data science code generation tasks. Tests capabilities in complex data processing, analysis, and programming across popular Python libraries used in data science workflows.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ds-arena-code:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ds-arena-code", "languages": [], "modality": "text", "name": "DS-Arena-Code", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.631, "raw_min": 0.631, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ds-arena-code:deepseek-v2.5", "reported_date": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500"}, "unit": null}, "slug": "llm-stats-ds-arena-code", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DeepSeek's internal Fill-in-the-Middle evaluation dataset for measuring code completion performance improvements in data science contexts", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ds-fim-eval:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ds-fim-eval", "languages": [], "modality": "text", "name": "DS-FIM-Eval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.783, "raw_min": 0.783, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ds-fim-eval:deepseek-v2.5", "reported_date": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500"}, "unit": null}, "slug": "llm-stats-ds-fim-eval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DSBench-FullStack is DeepSeek's internal full-stack development test set for evaluating coding agents on end-to-end software engineering tasks.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:dsbench-fullstack:deepseek-v4-flash-0731", "reported_at": "2026-07-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:dsbench-fullstack", "languages": [], "modality": "text", "name": "DSBench-FullStack", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.1, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.711, "raw_min": 0.687, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:dsbench-fullstack:deepseek-v4-pro-0813", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500"}, "unit": null}, "slug": "llm-stats-dsbench-fullstack", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DSBench-Hard is DeepSeek's internal test set of difficult coding-agent problems.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:dsbench-hard:deepseek-v4-flash-0731", "reported_at": "2026-07-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:dsbench-hard", "languages": [], "modality": "text", "name": "DSBench-Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.672, "raw_min": 0.596, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:dsbench-hard:deepseek-v4-pro-0813", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-dsbench-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "DUDE (Document Understanding Dataset and Evaluation) tests multi-page, multi-domain document understanding and reasoning.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:dude:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:dude", "languages": [], "modality": "multimodal", "name": "DUDE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.1, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.831, "raw_min": 0.828, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:dude:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500"}, "unit": null}, "slug": "llm-stats-dude", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "推理", "Reasoning", "代码", "Code", "Dynamic Benchmarking", "ICML 2025", "大语言模型", "LLM", "逻辑推理", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DyCodeEval introduces methods to generate dynamic evaluation dataset and metric. Using multi-agent cooperation to rewrite benchmarks at evaluation time, it generates semantically equivalent, diverse, and non-deterministic problems—reducing data contamination and enabling more trustworthy evaluation. DyCodeEval 提出了一种动态生成评测数据集和评测指标的方法。通过多智能体协作，在评测时对基准题目进行重写，生成语义等价、多样化且非确定性的问题，从而减少数据污染，实现更可信的评测。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1971", "languages": [], "modality": null, "name": "DyCodeEval", "openness": "unknown", "publisher": "Columbia University", "released": "2025-06-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1971-dycodeeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DyCodeEval", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multimodal mathematics and reasoning benchmark focused on dynamic visual problem solving.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:dynamath:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:dynamath", "languages": [], "modality": "multimodal", "name": "DynaMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.88, "raw_min": 0.681, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:dynamath:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500"}, "unit": null}, "slug": "llm-stats-dynamath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "数学", "Math", "多模态模型", "VLM", "逻辑推理", "数理能力", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "DynaMath is a dynamic visual math benchmark designed for in-depth assessment of VLMs. It includes 501 high-quality, multi-topic seed questions, each represented as a Python program enabling the automatic generation of a much larger set of concrete questions. DynaMath用于评估多模态大模型的数学能力，包括501个高质量、多主题的种子问题，每个问题都以Python程序表示，能够自动生成大量具体的多样化问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1374", "languages": [], "modality": "multimodal", "name": "DynaMath", "openness": "open", "publisher": "University of Illinois at Urbana-Champaign", "released": "2024-10-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1374-dynamath", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DynaMath", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "ACL 2024", "大语言模型", "LLM", "知识储备", "Knowledge", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "E-EVAL is the first comprehensive evaluation benchmark specifically tailored for Chinese K-12 education. E-EVAL comprises 4,351 multiple-choice questions spanning primary, middle, and high school levels, covering a diverse array of subjects. E-EVAL 是首个专门针对中国 K-12 教育的综合评估基准。E-EVAL 包含 4,351 道选择题，涵盖小学、初中和高中各个年级，涉及多种学科。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1080", "languages": ["Chinese"], "modality": null, "name": "E-EVAL", "openness": "open", "publisher": "University of Science and Technology of China", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1080-e-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/E-EVAL", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multilingual closed-book question answering dataset that evaluates cross-lingual knowledge transfer in large language models across 12 languages, using knowledge-seeking questions based on Wikipedia articles that exist only in one language", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:eclektic:gemma-3-12b-it", "reported_at": "2025-03-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:eclektic", "languages": [], "modality": "text", "name": "ECLeKTic", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 19.0, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.19, "raw_min": 0.014, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:eclektic:gemma-3n-e4b-it", "reported_date": "2025-06-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500"}, "unit": null}, "slug": "llm-stats-eclektic", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500", "unit": null}, {"aliases": [], "categories": ["创作", "Creation", "指令跟随", "Instruct", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "EditInspector is a novel benchmark for evaluation of text-guided image edits, based on human annotations collected using an extensive template for edit verification. We leverage EditInspector to evaluate the performance of state-of-the-art (SoTA) vision and language models. EditInspector 评估最先进（SoTA）视觉和语言模型在多个维度上评估编辑的性能，包括准确性、瑕疵检测、视觉质量、与图像场景的无缝融合、遵循常识以及描述编辑引起变化的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1970", "languages": [], "modality": null, "name": "EditInspector", "openness": "unknown", "publisher": "The Hebrew University of Jerusalem , TelAviv University , Google Research", "released": "2025-06-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1970-editinspector", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EditInspector", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "推理", "Reasoning", "多模态模型", "VLM", "视频理解", "Video Understanding", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "EgoNormia is a challenging QA benchmark that tests VLMs' ability to reason over norms in context. The datset consists of 1,853 physically grounded egocentric interaction clips from Ego4D and corresponding five-way multiple-choice questions tasks for each. EgoNormia 是一个具有挑战性的问答基准，用于测试 VLMs 在上下文中推理规范的能力。该数据集包含来自 Ego4D 的 1,853 个物理基础化的以自我为中心的交互剪辑，以及每个剪辑对应的五选一多项选择题任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1604", "languages": [], "modality": null, "name": "EgoNormia", "openness": "unknown", "publisher": "University of Arizona, Stanford University, etc.", "released": "2025-02-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1604-egonormia", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EgoNormia", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A diagnostic benchmark for very long-form video language understanding consisting of over 5000 human curated multiple choice questions based on 3-minute video clips from Ego4D, covering a broad range of natural human activities and behaviors", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:egoschema:gemini-1.0-pro", "reported_at": "2024-02-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:egoschema", "languages": [], "modality": "video", "name": "EgoSchema", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.9, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.779, "raw_min": 0.557, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:egoschema:qwen2-vl-72b", "reported_date": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500"}, "unit": null}, "slug": "llm-stats-egoschema", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "NeurIPS 2024", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "EHRNoteQA evaluates LLM's ability to assist clinical decision-making based on electronic health records, comprising 962 different QA pairs each linked to distinct patients' discharge summaries. EHRNoteQA用于评估LLM基于电子健康记录辅助临床决策的能力，由962个问答对组成，每个问答对都与不同患者的出院总结相关联。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1327", "languages": [], "modality": null, "name": "EHRNoteQA", "openness": "unknown", "publisher": "KAIST", "released": "2024-02-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1327-ehrnoteqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EHRNoteQA", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "学科", "Examination", "安全", "Safety", "教育", "多模态模型", "开源收录", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ELBench is a multidimensional benchmark for education-facing large language models. It covers safety and trustworthiness, high-level educational cultivation, and  general capabilities, combining objective evaluation with LLM-as-a-Judge. ELBench 是面向教育场景大语言模型的多维评测集，覆盖安全可信、高阶育人和通用能力。公开任务包含安全回答、教育开放任务、选择题、数学与推理任务，并结合客观规则和大模型判决进行评测。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:2571", "languages": ["English", "Chinese"], "modality": null, "name": "ELBench", "openness": "open", "publisher": "华东师范大学", "released": "2026-06-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2571-elbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ELBench", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "Embodied Decision Making", "NeurIPS 2024", "物理智能", "Embodied AI", "逻辑推理", "具身交互", "Embodied Interaction", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Embodied Agent Interface supports the formalization of various types of tasks and input-output specifications of LLM-based modules, offering a comprehensive assessment of LLMs' performance for different subtasks and pinpointing the strengths and weaknesses in LLM-powered embodied AI systems. Embodied Agent Interface支持各种类型的任务和基于LLM模块的输入输出规范的形式化，对LLM在不同子任务中的性能进行了全面评估，指出了基于LLM的具身AI系统的优势和劣势。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1243", "languages": [], "modality": null, "name": "EmbodiedAgentInterface", "openness": "open", "publisher": "Stanford University", "released": "2024-06-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1243-embodiedagentinterface", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedAgentInterface", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "智能体", "Agent", "物理智能", "Embodied AI", "跨模态推理", "Cross-modal Reasoning", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "EmbodiedBench is a comprehensive benchmark designed to evaluate Multi-modal Large Language Models (MLLMs) as embodied agents. EmbodiedBench，这是一个旨在评估多模态大型语言模型作为具身智能体的全面基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1564", "languages": [], "modality": "multimodal", "name": "EmbodiedBench", "openness": "unknown", "publisher": "University of Illinois Urbana-Champaign, etc.", "released": "2025-02-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1564-embodiedbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedBench", "unit": null}, {"aliases": [], "categories": ["spatial_reasoning", "embodied", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "EmbSpatialBench evaluates embodied spatial understanding and reasoning capabilities.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:embspatialbench:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:embspatialbench", "languages": [], "modality": "image", "name": "EmbSpatialBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.846, "display_multiplier": 1, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.846, "raw_min": 0.825, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:embspatialbench:qwen3.6-27b", "reported_date": "2026-04-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500"}, "unit": null}, "slug": "llm-stats-embspatialbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "EMMA (Enhanced MultiModal reAsoning) is a benchmark for organic multimodal reasoning across mathematics, physics, chemistry, and coding.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:emma:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:emma", "languages": [], "modality": "multimodal", "name": "EMMA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 79.3, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.793, "raw_min": 0.784, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:emma:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500"}, "unit": null}, "slug": "llm-stats-emma", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "EMMA is composed of 2,788 problems, of which 1,796 are newly constructed, across four domains. Within each subject, we further provide fine-grained labels for each question based on the specific skills it measures. EMMA 由 2,788 个问题组成，其中 1,796 个是新构建的，涵盖四个领域。在每个主题中，我们根据所测量的具体技能为每个问题提供细粒度标签。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1640", "languages": [], "modality": "multimodal", "name": "EMMA", "openness": "restricted", "publisher": "UESTC, SYSU, etc.", "released": "2025-01-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1640-emma", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EMMA", "unit": null}, {"aliases": [], "categories": ["agentic", "business"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic business operations", "evidence_summary": {"document_count": 1, "model_count": 33, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:enterpriseops-gym-aa:f0083258-8646-45b8-8082-7aaf6c2ea82a", "reported_at": "2025-08-05", "source_url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:enterpriseops-gym-aa", "languages": [], "modality": null, "name": "EnterpriseOps-Gym-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 33, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.11906893464637, "display_multiplier": 100, "model_count": 33, "model_count_basis": "source_model_id", "numeric_count": 33, "raw_max": 0.5111906893464637, "raw_min": 0.25544613548194567, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:enterpriseops-gym-aa:cd55210d-358e-4df1-ba9c-9acb5f186cc9", "reported_date": "2026-06-09", "source_url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa"}, "unit": null}, "slug": "artificial-analysis-enterpriseops-gym-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "(E-commerce Product Review Dataset for Sentiment Analysis), also known as EPRSTMT, is a binary sentiment analysis dataset based on product reviews on e-commerce platform. Each sample is labelled as Positive or Negative. It collect by ICIP Lab of Beijing Normal University. EPRSTMT，也称作电子商务产品评论情感分析数据集，是一个基于电子商务平台上的产品评论的二元情感分析数据集。每个样本都被标记为积极或消极。该数据集由北京师范大学 ICIP 实验室收集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:522", "languages": [], "modality": null, "name": "EPRSTMT", "openness": "unknown", "publisher": null, "released": "2021-07-15", "released_reference": {"basis": "paper_first_version", "note": "The FewCLUE introduction includes its new EPRSTMT product-review sentiment task.", "source_key": "opencompass:522", "source_url": "https://arxiv.org/abs/2107.07498"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-522-eprstmt", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EPRSTMT", "unit": null}, {"aliases": [], "categories": ["reasoning", "roleplay", "general", "creativity", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "EQ-Bench is an LLM-judged test evaluating active emotional intelligence abilities, understanding, insight, empathy, and interpersonal skills. The test set contains 45 challenging roleplay scenarios, most of which constitute pre-written prompts spanning 3 turns. The benchmark evaluates the performance of models by validating responses against several criteria and conducts pairwise comparisons to report a normalized Elo computation for each model.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:eq-bench:grok-4.1-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:eq-bench", "languages": [], "modality": "text", "name": "EQ-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 1586.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 1586.0, "raw_min": 1585.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:eq-bench:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-eq-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "Medical", "科学智能", "AI for Science", "逻辑推理", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ER-Reason is a large-scale benchmark suite for evaluating the clinical reasoning capabilities of large language models (LLMs) in the emergency room (ER) — a high-stakes environment where clinicians make rapid, life-critical decisions. ER-Reason 是一个大规模基准套件，用于评估大语言模型（LLMs）在急诊室（ER）中的临床推理能力。急诊室是一个高风险环境，临床医生需要快速做出关乎生命的关键决策。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1892", "languages": [], "modality": null, "name": "ER-Reason", "openness": "unknown", "publisher": "University of California, Berkeley,etc.", "released": "2025-05-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1892-er-reason", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ER-Reason", "unit": null}, {"aliases": [], "categories": ["reasoning", "spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Embodied Reasoning Question Answering benchmark consisting of 400 multiple-choice visual questions across spatial reasoning, trajectory reasoning, action reasoning, state estimation, and multi-view reasoning for evaluating AI capabilities in physical world interactions", "evidence_summary": {"document_count": 1, "model_count": 24, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:erqa:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:erqa", "languages": [], "modality": "multimodal", "name": "ERQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 24, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.8, "display_multiplier": 100, "model_count": 24, "model_count_basis": "source_model_id", "numeric_count": 24, "raw_max": 0.778, "raw_min": 0.352, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:erqa:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500"}, "unit": null}, "slug": "llm-stats-erqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A rigorous code synthesis evaluation framework that augments existing datasets with extensive test cases generated by LLM and mutation-based strategies to better assess functional correctness of generated code, including HumanEval+ with 80x more test cases", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:evalplus:qwen2-72b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:evalplus", "languages": [], "modality": "text", "name": "EvalPlus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.803, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.803, "raw_min": 0.703, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:evalplus:kimi-k2-base", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500"}, "unit": null}, "slug": "llm-stats-evalplus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "智能体", "Agent", "指令跟随", "Instruct", "worldmodel", "embodied AI", "manipulate", "物理智能", "指令遵循", "Instruction Following", "具身交互", "Embodied Interaction", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "EWMBench is a benchmark for evaluating Embodied World Models, covering aspects such as scene consistency, motion correctness, and semantic alignment. It includes a diverse dataset and multi-dimensional metrics tailored for embodied manipulation tasks. EWMBench 是一个用于评估具身世界模型的基准，涵盖“场景一致性”、“动作正确性”和“语义对齐”等方面。它包含多样化的数据集和面向具身任务的多维评估指标，能够揭示现有模型的局限，并为生成具物理基础、任务导向的视频提供评价标准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1843", "languages": [], "modality": "multimodal", "name": "EWMBench", "openness": "restricted", "publisher": "Agibot", "released": "2025-05-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1843-ewmbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EWMBench", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ExploitBench is a cybersecurity benchmark that evaluates a model's ability to discover and exploit software vulnerabilities, reported as the fraction of challenges where the model captures the target (Cap%).", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:exploitbench:claude-fable-5", "reported_at": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:exploitbench", "languages": [], "modality": "text", "name": "ExploitBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.0, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.78, "raw_min": 0.332, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:exploitbench:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500"}, "unit": null}, "slug": "llm-stats-exploitbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500", "unit": null}, {"aliases": ["ExploitBench", "ExploitBench (Cap%)"], "categories": ["security"], "collected_at": null, "description": "Offensive-security capability measured as a risk indicator against a preparedness threshold, not a leaderboard to top. Reported as a capability percentage. Published by Seunghyun Lee and David Brumley (Carnegie Mellon; Brumley also lists Bugcrowd) as arXiv:2605.14153 with code at github.com/exploitbench/exploitbench. 41 V8 N-day vulnerabilities scored by 16 grader-verified capability flags; vendor runs report the capability percentage without publishing per-flag results, so a reported number is not reproducible from the paper alone.", "evidence_summary": {"document_count": 2, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:exploitbench", "languages": [], "modality": null, "name": "ExploitBench", "openness": "unknown", "publisher": null, "released": "2026-05-13", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:exploitbench", "source_url": "https://exploitbench.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "exploitbench", "source": "model_reports", "source_url": "https://exploitbench.ai/", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ExploitGym is a large-scale, realistic benchmark built from real-world vulnerabilities across userspace programs, Google's V8 engine, and the Linux kernel. Given a vulnerability and a proof-of-vulnerability input, agents must craft a working end-to-end exploit that achieves unauthorized code execution.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:exploitgym:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:exploitgym", "languages": [], "modality": "text", "name": "ExploitGym", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 33.7, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.337, "raw_min": 0.124, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:exploitgym:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500"}, "unit": null}, "slug": "llm-stats-exploitgym", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "factuality", "grounding"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark evaluating language models' ability to generate factually accurate and well-grounded responses based on long-form input context, comprising 1,719 examples with documents up to 32k tokens requiring detailed responses that are fully grounded in provided documents", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:facts-grounding:gemini-2.0-flash", "reported_at": "2024-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:facts-grounding", "languages": [], "modality": "text", "name": "FACTS Grounding", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.8, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.878, "raw_min": 0.364, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:facts-grounding:gemini-2.5-pro-preview-06-05", "reported_date": "2025-06-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500"}, "unit": null}, "slug": "llm-stats-facts-grounding", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A fine-grained atomic evaluation metric for factual precision in long-form text generation that breaks generated text into atomic facts and computes the percentage supported by reliable knowledge sources, with automated assessment using retrieval and language models", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:factscore:gpt-5-2025-08-07", "reported_at": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:factscore", "languages": [], "modality": "text", "name": "FActScore", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.03, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.9703, "raw_min": 0.01, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:factscore:grok-4.1-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500"}, "unit": null}, "slug": "llm-stats-factscore", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "安全", "Safety", "大语言模型", "LLM", "逻辑推理", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Safety alignment approaches in large language models (LLMs) often lead\nto the over-refusal of benign queries, significantly diminishing their utility\nin sensitive scenarios. To address this challenge, we introduce FalseReject,\na comprehensive resource containing 16k seemingly toxic queries accompani FalseReject 构建了包含 16 000 条表面“有毒”但实为良性的查询样本，覆盖 44 个安全相关类别，并提出一种基于图信息的对抗多智能体交互框架，用以生成多样且复杂的提示–响应对，并在响应中引入显式推理链，帮助模型更准确地区分安全与不安全上下文；该工作还为标准指令调优模型和推理导向模型分别准备了专项训练集，并附带人工标注的基准测试集，针对 29 款最先进 LLM 进行了大规模评估，结果表明经 FalseReject 监督微调后，模型在显著减少对良性查询的过度拒绝的同时，不仅未损失整体安全性，也保持了语言生成能力，为敏感场景下提升 LLM 可用性提供了首个系统化、可复现的资源与方法框", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1980", "languages": ["English"], "modality": null, "name": "FalseReject", "openness": "unknown", "publisher": "Amazon", "released": "2025-05-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1980-falsereject", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FalseReject", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "FEA基准测试", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "FEABench is a benchmark to evaluate the ability of large language models (LLMs) and LLM agents to simulate and solve physics, mathematics and engineering problems using finite element analysis (FEA). FEABench，一个用于评估大型语言模型和LLM代理使用有限元分析（FEA）模拟和解决物理、数学及工程问题能力的基准测试。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1733", "languages": [], "modality": null, "name": "FEABench", "openness": "unknown", "publisher": "Google Research，Harvard University", "released": "2025-04-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1733-feabench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FEABench", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "FedMABench is an open-source benchmark for federated training and evaluation of mobile agents, specifically designed for heterogeneous scenarios. FedMABench 是一个开源的联邦训练和评估移动代理的基准，特别为异构场景设计。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1618", "languages": [], "modality": null, "name": "FedMABench", "openness": "unknown", "publisher": "ZJU, SJTU, etc.", "released": "2025-03-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1618-fedmabench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FedMABench", "unit": null}, {"aliases": [], "categories": ["safety", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FigQA is a multiple-choice benchmark on interpreting scientific figures from biology papers. It evaluates dual-use biological knowledge and multimodal reasoning relevant to bioweapons development.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:figqa:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:figqa", "languages": [], "modality": "multimodal", "name": "FigQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.89, "raw_min": 0.34, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:figqa:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500"}, "unit": null}, "slug": "llm-stats-figqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Ant Group and Shanghai University of Finance and Economics jointly launched the financial benchmark，Fin-Eva Version 1.0, covering multiple financial scenarios and subjects such as wealth management, insurance, investment research. The number of this benchmark's questions reaches 13,000+. 蚂蚁集团、上海财经大学联合推出金融评测集Fin-Eva Version 1.0，覆盖财富管理、保险、投资研究等多个金融场景以及金融专业主题学科，总评测题数目达到13,000+。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:895", "languages": [], "modality": null, "name": "Fin-Eva", "openness": "unknown", "publisher": null, "released": "2023-12-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-895-fin-eva", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Fin-Eva", "unit": null}, {"aliases": [], "categories": ["reasoning", "finance", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Finance Agent is a benchmark for evaluating AI models on agentic financial analysis tasks, testing their ability to process financial data, perform calculations, and generate accurate analyses across various financial domains.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:finance-agent:claude-opus-4-6", "reported_at": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:finance-agent", "languages": [], "modality": "text", "name": "Finance Agent", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.4, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.644, "raw_min": 0.537, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:finance-agent:claude-opus-4-7", "reported_date": "2026-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500"}, "unit": null}, "slug": "llm-stats-finance-agent", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "finance", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Finance Agent v1.1 is an agentic financial-analysis benchmark that evaluates models on real-world finance workflows, including retrieving and reasoning over financial documents and performing multi-step calculations.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:finance-agent-v1.1:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:finance-agent-v1.1", "languages": [], "modality": "text", "name": "Finance Agent v1.1", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.699999999999996, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.607, "raw_min": 0.56, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:finance-agent-v1.1:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500"}, "unit": null}, "slug": "llm-stats-finance-agent-v1-1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "finance", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Finance Agent v2 is an agentic financial-analysis benchmark from Vals that evaluates models on real-world finance workflows, measuring their ability to retrieve and reason over financial documents, perform multi-step calculations, and produce accurate analyses.", "evidence_summary": {"document_count": 1, "model_count": 26, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:finance-agent-v2:claude-haiku-4-5-20251001", "reported_at": "2025-10-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:finance-agent-v2", "languages": [], "modality": "text", "name": "Finance Agent v2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 26, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.86, "display_multiplier": 100, "model_count": 26, "model_count_basis": "source_model_id", "numeric_count": 26, "raw_max": 0.5786, "raw_min": 0.2789, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:finance-agent-v2:gemini-3.5-flash", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-finance-agent-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "finance", "economics"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A large-scale dataset for numerical reasoning over financial data with question-answering pairs written by financial experts, featuring complex numerical reasoning and understanding of heterogeneous representations with annotated gold reasoning programs for full explainability", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:finqa:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:finqa", "languages": [], "modality": "text", "name": "FinQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.772, "raw_min": 0.652, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:finqa:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500"}, "unit": null}, "slug": "llm-stats-finqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "finance", "economics"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FinSearchComp T2&T3 is a combined benchmark for evaluating financial search and reasoning capabilities on Tier 2 and Tier 3 tasks, testing models' ability to retrieve and analyze complex financial information using tools.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:finsearchcomp-t2-t3:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:finsearchcomp-t2-t3", "languages": [], "modality": "text", "name": "FinSearchComp T2&T3", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.80000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.678, "raw_min": 0.678, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:finsearchcomp-t2-t3:kimi-k2.5", "reported_date": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500"}, "unit": null}, "slug": "llm-stats-finsearchcomp-t2-t3", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "finance", "economics"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FinSearchComp-T3 is a benchmark for evaluating financial search and reasoning capabilities, testing models' ability to retrieve and analyze financial information using tools.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:finsearchcomp-t3:kimi-k2-thinking-0905", "reported_at": "2025-09-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:finsearchcomp-t3", "languages": [], "modality": "text", "name": "FinSearchComp-T3", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 47.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.474, "raw_min": 0.474, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:finsearchcomp-t3:kimi-k2-thinking-0905", "reported_date": "2025-09-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500"}, "unit": null}, "slug": "llm-stats-finsearchcomp-t3", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Flame-VLM-Code evaluates multimodal models on visual code generation tasks, measuring ability to generate code from visual inputs such as UI mockups and design specifications.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:flame-vlm-code:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:flame-vlm-code", "languages": [], "modality": "multimodal", "name": "Flame-VLM-Code", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.938, "raw_min": 0.938, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:flame-vlm-code:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500"}, "unit": null}, "slug": "llm-stats-flame-vlm-code", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "NAACL 2024", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Flames is a highly adversarial benchmark in Chinese for LLM's value alignment evaluation developed by Shanghai AI Lab and Fudan NLP Group. Flames meticulously designs a dataset of 2,251 highly adversarial, manually crafted prompts, each tailored to probe a specific value dimension (i.e., Fairness, Safety, Morality, Legality, Data protection). Currently,  Flames releases 1,000 prompts for public use (Flames_1k_Chinese). Flames 是上海人工智能实验室和复旦大学 NLP团队开发的 LLM 价值对齐方向的中文高度对抗性基准。Flames 精心设计了一个由 2,251 个高度对抗性、人工创建的提示词成的评测集，每个提示词都经过精心设计，以探究特定的价值维度（即公平、安全、道德、合法、数据保护）。目前，Flames 发布了 1,000 个提示词供公众使用（Flames_1k_Chinese）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:945", "languages": ["Chinese"], "modality": null, "name": "Flames", "openness": "restricted", "publisher": null, "released": "2024-03-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-945-flames", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Flames", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Flexible Length Question Answering dataset for evaluating the impact of input length on reasoning performance of language models, featuring True/False questions embedded in contexts of varying lengths (250-3000 tokens) across three reasoning tasks: Monotone Relations, People In Rooms, and simplified Ruletaker", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:flenqa:phi-4-reasoning", "reported_at": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:flenqa", "languages": [], "modality": "text", "name": "FlenQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.89999999999999, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.979, "raw_min": 0.977, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:flenqa:phi-4-reasoning-plus", "reported_date": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500"}, "unit": null}, "slug": "llm-stats-flenqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["speech_to_text", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Few-shot Learning Evaluation of Universal Representations of Speech - a parallel speech dataset in 102 languages built on FLoRes-101 with approximately 12 hours of speech supervision per language for tasks including ASR, speech language identification, translation and retrieval. Scores are shown as speech recognition accuracy (1 - word error rate), so higher is better.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:fleurs:gemini-1.0-pro", "reported_at": "2024-02-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:fleurs", "languages": [], "modality": "audio", "name": "FLEURS", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.89999999999999, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.959, "raw_min": 0.864, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:fleurs:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500"}, "unit": null}, "slug": "llm-stats-fleurs", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Flores is a benchmark dataset for machine translation between English and low-resource languages, which consists of sentences translated from Wikipedia, involving English and four low-resource languages, namely Nepali, Sinhala, Khmer and Pashto. Flores has two versions, we use the Flores-101 version here, which is the second version including 101 languages besides english. Flores是一个用于评估低资源语言机器翻译的基准数据集，它包含了从维基百科翻译的句子，涉及英语和四种低资源语言，分别是尼泊尔语、僧伽罗语、高棉语和普什图语。Flores有两个版本，我们这里使用的是第一个版本Flores-101，它包含有除英语外的101种语言。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:509", "languages": ["English"], "modality": null, "name": "Flores", "openness": "unknown", "publisher": null, "released": "2021-06-06", "released_reference": {"basis": "paper_first_version", "note": "The Hub record explicitly links the FLORES-101 introduction; its selected languages are contained in that benchmark.", "source_key": "opencompass:509", "source_url": "https://arxiv.org/abs/2106.03193"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-509-flores", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Flores", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "FLUB evaluates the reasoning and understanding abilities of LLMs. It includes three tasks with increasing difficulty, consisting of the tricky, humorous, and misleading texts collected from the real internet environment. FLUB用于评估LLM的推理和理解能力，其中包含3个难度递进的任务，由从真实互联网环境中收集的狡猾、幽默和误导性的文本构成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1324", "languages": [], "modality": null, "name": "FLUB", "openness": "unknown", "publisher": "Tsinghua University", "released": "2024-02-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1324-flub", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FLUB", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "安全", "Safety", "Forgery Detection", "Large Vision Language Models", "多模态模型", "VLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A Comprehensive Forgery Detection Benchmark Suite for Large Vision Language Models A Comprehensive Forgery Detection Benchmark Suite for Large Vision Language Models", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1671", "languages": [], "modality": "multimodal", "name": "Forensics-bench", "openness": "restricted", "publisher": "The University of Hong Kong", "released": "2025-03-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1671-forensics-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Forensics-bench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "AVQA", "多模态模型", "VLM", "逻辑推理", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "FortisAVQA is the first dataset designed to assess the robustness of AVQA models. FortisAVQA，这是首个用于评估AVQA模型鲁棒性的数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1737", "languages": [], "modality": "multimodal", "name": "FortisAVQA", "openness": "unknown", "publisher": "Hong Kong University，etc.", "released": "2025-04-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1737-fortisavqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FortisAVQA", "unit": null}, {"aliases": [], "categories": ["reasoning", "search"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Factuality, Retrieval, And reasoning MEasurement Set - a unified evaluation dataset of 824 challenging multi-hop questions for testing retrieval-augmented generation systems across factuality, retrieval accuracy, and reasoning capabilities, requiring integration of 2-15 Wikipedia articles per question", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frames:deepseek-v3", "reported_at": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frames", "languages": [], "modality": "text", "name": "FRAMES", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.87, "raw_min": 0.733, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frames:kimi-k2-thinking-0905", "reported_date": "2025-09-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500"}, "unit": null}, "slug": "llm-stats-frames", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "NAACL 2024", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "FREB-TQA is a Fine-grained Robustness Evaluation Benchmark for Table Question Answering. FREB-TQA 是一个细粒度的稳健性评估基准，专注于表格问答（TQA）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1145", "languages": [], "modality": null, "name": "FREB-TQA", "openness": "unknown", "publisher": "Bosch Center for Artificial Intelligence", "released": "2024-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1145-freb-tqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FREB-TQA", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "French version of MMLU-Pro, a multilingual benchmark for evaluating language models' cross-lingual reasoning capabilities across 14 diverse domains including mathematics, physics, chemistry, law, engineering, psychology, and health.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:french-mmlu:ministral-8b-instruct-2410", "reported_at": "2024-10-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:french-mmlu", "languages": [], "modality": "text", "name": "French MMLU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.49999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.575, "raw_min": 0.575, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:french-mmlu:ministral-8b-instruct-2410", "reported_date": "2024-10-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500"}, "unit": null}, "slug": "llm-stats-french-mmlu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Frontier Science is a benchmark of exceptionally challenging scientific reasoning problems spanning advanced natural-science domains, designed to test expert-level scientific understanding and multi-step reasoning.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontier-science:mai-code-1-flash", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontier-science", "languages": [], "modality": "text", "name": "Frontier Science", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 58.199999999999996, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.582, "raw_min": 0.582, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontier-science:mai-code-1-flash", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500"}, "unit": null}, "slug": "llm-stats-frontier-science", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Frontier-Bench v0.1 evaluates agentic terminal coding. Anthropic reports results using the mini-SWE-agent harness and a GKE backend, measured as mean reward across five attempts per task.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontier-bench-v0.1:claude-opus-5", "reported_at": "2026-07-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontier-bench-v0.1", "languages": [], "modality": "text", "name": "Frontier-Bench v0.1", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 43.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.433, "raw_min": 0.433, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontier-bench-v0.1:claude-opus-5", "reported_date": "2026-07-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500"}, "unit": null}, "slug": "llm-stats-frontier-bench-v0-1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500", "unit": null}, {"aliases": ["FrontierChallenge", "Frontier Challenge"], "categories": ["scientific_agent"], "collected_at": null, "description": "Pass-rate results are model-scaffold measurements over the 97 released tasks with one trajectory per system-task pair, so a reported percentage is not a model-only number and does not estimate reliability beyond this release or under a different scaffold.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000frontier_challenge\u0000frontier_challenge_leaderboard_2026_08_25\u0000frontier_challenge_97_tasks_2026_08_25\u000097 released tasks; one trajectory per system-task pair; pass = native Score >= 99.9; agent scaffold: Claude Code\u0000Apodex 1.1", "reported_at": "2026-08-25", "source_url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/"}, "first_score_reported_at": "2026-08-25", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000frontier_challenge\u0000frontier_challenge_leaderboard_2026_08_25\u0000frontier_challenge_97_tasks_2026_08_25\u000097 released tasks; one trajectory per system-task pair; pass = native Score >= 99.9; agent scaffold: Claude Code\u0000Apodex 1.1", "reported_at": "2026-08-25", "source_url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:frontier_challenge", "languages": [], "modality": null, "name": "FrontierChallenge", "openness": "unknown", "publisher": null, "released": "2026-08-25", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:frontier_challenge", "source_url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 20.6, "display_multiplier": 1, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 20.6, "raw_min": 3.1, "source_reference": {"obs_id": "curated\u0000frontier_challenge\u0000frontier_challenge_leaderboard_2026_08_25\u0000frontier_challenge_97_tasks_2026_08_25\u000097 released tasks; one trajectory per system-task pair; pass = native Score >= 99.9; agent scaffold: Claude Code\u0000Grok 4.6", "observation_id": "curated\u0000frontier_challenge\u0000frontier_challenge_leaderboard_2026_08_25\u0000frontier_challenge_97_tasks_2026_08_25\u000097 released tasks; one trajectory per system-task pair; pass = native Score >= 99.9; agent scaffold: Claude Code\u0000Grok 4.6", "reported_at": "2026-08-25", "reported_date": "2026-08-25", "source_id": "frontier_challenge_leaderboard_2026_08_25", "source_url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/"}, "unit": "percent"}, "slug": "frontier_challenge", "source": "model_reports", "source_url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierCode is Cognition's coding evaluation that tests whether models can pass difficult coding tasks while meeting the standards of high-quality production codebases. The Diamond subset contains the hardest problems.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontiercode:claude-fable-5", "reported_at": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontiercode", "languages": [], "modality": "text", "name": "FrontierCode", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.3, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.613, "raw_min": 0.388, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontiercode:grok-4.6", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500"}, "unit": null}, "slug": "llm-stats-frontiercode", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500", "unit": null}, {"aliases": ["FrontierCode", "FrontierCode (Diamond)"], "categories": ["coding_agent"], "collected_at": null, "description": "Cognition's production-standard coding evaluation, asking whether models write *good* code rather than merely correct code. Unrelated to Terminal-Bench's Frontier-Bench despite the shared prefix. Reported per reasoning-effort setting (the Diamond subset at xhigh), so the effort level has to travel with the number. First-party to a competing coding agent's vendor, and revised as 1.1 on 2026-07-07, so the methodology version has to travel with the number too.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:frontiercode", "languages": [], "modality": null, "name": "FrontierCode", "openness": "unknown", "publisher": null, "released": "2026-06-08", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:frontiercode", "source_url": "https://cognition.com/blog/frontier-code"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "frontiercode", "source": "model_reports", "source_url": "https://cognition.com/blog/frontier-code", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierCode 1.1 evaluates whether coding-agent changes are mergeable, using unit tests, maintainer-defined rubrics, and verifiers. Runs flagged for unfair internet use receive a zero score.", "evidence_summary": {"document_count": 1, "model_count": 16, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontiercode-1.1:claude-opus-4-7", "reported_at": "2026-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontiercode-1.1", "languages": [], "modality": "text", "name": "FrontierCode 1.1", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 16, "score_direction": "higher_is_better", "score_summary": {"display_max": 53.5, "display_multiplier": 100, "model_count": 16, "model_count_basis": "source_model_id", "numeric_count": 16, "raw_max": 0.535, "raw_min": 0.102, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontiercode-1.1:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500"}, "unit": null}, "slug": "llm-stats-frontiercode-1-1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierCS is a benchmark of frontier computer-science problems requiring deep theoretical understanding and rigorous multi-step reasoning at the edge of the field.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontiercs:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontiercs", "languages": [], "modality": "text", "name": "FrontierCS", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 50.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.508, "raw_min": 0.463, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontiercs:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500"}, "unit": null}, "slug": "llm-stats-frontiercs", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark of hundreds of original, exceptionally challenging mathematics problems crafted and vetted by expert mathematicians, covering most major branches of modern mathematics from number theory and real analysis to algebraic geometry and category theory.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontiermath:o1-2024-12-17", "reported_at": "2024-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontiermath", "languages": [], "modality": "text", "name": "FrontierMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.0, "display_multiplier": 100, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 0.89, "raw_min": 0.055, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontiermath:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500"}, "unit": null}, "slug": "llm-stats-frontiermath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierMath Tier 4 subset from the v2 evaluation release.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontiermath-tier-4-v2:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontiermath-tier-4-v2", "languages": [], "modality": "text", "name": "FrontierMath Tier 4 (v2)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.83, "raw_min": 0.585, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontiermath-tier-4-v2:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-frontiermath-tier-4-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "science"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierScience Olympiad is a benchmark of olympiad-level scientific reasoning problems, testing expert understanding and multi-step reasoning across advanced natural-science domains.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontierscience-olympiad:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontierscience-olympiad", "languages": [], "modality": "text", "name": "FrontierScience Olympiad", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.76, "raw_min": 0.748, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontierscience-olympiad:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500"}, "unit": null}, "slug": "llm-stats-frontierscience-olympiad", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "science"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierScience Research is a benchmark evaluating AI models on cutting-edge scientific research questions requiring deep domain expertise, multi-step reasoning, and synthesis of complex scientific concepts across disciplines.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontierscience-research:muse-spark", "reported_at": "2026-04-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontierscience-research", "languages": [], "modality": "text", "name": "FrontierScience Research", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 38.3, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.383, "raw_min": 0.213, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontierscience-research:muse-spark", "reported_date": "2026-04-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500"}, "unit": null}, "slug": "llm-stats-frontierscience-research", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierSWE measures whether an agent can complete open-ended technical projects at the scale of hours to tens of hours, spanning systems optimization, large-scale code construction, and applied ML research. Performance is reported as a dominance score, where higher is better.", "evidence_summary": {"document_count": 1, "model_count": 16, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontierswe:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontierswe", "languages": [], "modality": "text", "name": "FrontierSWE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 16, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.0, "display_multiplier": 100, "model_count": 16, "model_count_basis": "source_model_id", "numeric_count": 16, "raw_max": 0.9, "raw_min": 0.22, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontierswe:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500"}, "unit": null}, "slug": "llm-stats-frontierswe", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "FrontierSWE (Impl.) evaluates software engineering implementation ability and reports model ranking on implementation tasks. Lower rank is better.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:frontier-swe-impl:mimo-v2.5-pro", "reported_at": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:frontier-swe-impl", "languages": [], "modality": "text", "name": "FrontierSWE (Impl.)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 3.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 3.4, "raw_min": 3.4, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:frontier-swe-impl:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500"}, "unit": null}, "slug": "llm-stats-frontier-swe-impl", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "English subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:fullstackbench-en:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:fullstackbench-en", "languages": [], "modality": "text", "name": "FullStackBench en", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.6, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.626, "raw_min": 0.581, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:fullstackbench-en:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500"}, "unit": null}, "slug": "llm-stats-fullstackbench-en", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Chinese subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:fullstackbench-zh:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:fullstackbench-zh", "languages": [], "modality": "text", "name": "FullStackBench zh", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 58.699999999999996, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.587, "raw_min": 0.55, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:fullstackbench-zh:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500"}, "unit": null}, "slug": "llm-stats-fullstackbench-zh", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A functional variant of the MATH benchmark that tests language models' ability to generalize reasoning patterns across different problem instances, revealing the reasoning gap between static and functional performance.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:functionalmath:gemini-1.5-flash", "reported_at": "2024-05-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:functionalmath", "languages": [], "modality": "text", "name": "FunctionalMATH", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.60000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.646, "raw_min": 0.536, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:functionalmath:gemini-1.5-pro", "reported_date": "2024-05-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500"}, "unit": null}, "slug": "llm-stats-functionalmath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GAIA is a benchmark which aims at evaluating next-generation LLMs (LLMs with augmented capabilities due to added tooling, efficient prompting, access to search, etc), mading of more than 450 non-trivial question with an unambiguous answer, requiring different levels of tooling and autonomy to solve. GAIA 是一个旨在评估下一代LLMs（由于增加了工具、高效的提示、访问搜索等功能而具有增强能力的LLMs）的基准，由超过 450 个非平凡问题组成，这些问题有明确的答案，需要不同层次的工具和自主性来解决。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1599", "languages": [], "modality": null, "name": "GAIA", "openness": "restricted", "publisher": "Meta, Huggingface, AutoGPT", "released": "2023-11-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1599-gaia", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAIA", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GAIA2 evaluates general-purpose AI agents on real-world, multi-step questions that require reasoning, tool use, and information retrieval.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gaia2:muse-glimmer-30b", "reported_at": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gaia2", "languages": [], "modality": "multimodal", "name": "GAIA2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 43.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.433, "raw_min": 0.433, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gaia2:muse-glimmer-30b", "reported_date": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500"}, "unit": null}, "slug": "llm-stats-gaia2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GameWorld evaluates agents on interactive game environments, testing perception, planning, and sequential decision-making to accomplish in-game objectives.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gameworld:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gameworld", "languages": [], "modality": "multimodal", "name": "GameWorld", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 31.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.312, "raw_min": 0.259, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gameworld:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500"}, "unit": null}, "slug": "llm-stats-gameworld", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GAOKAO-bench is an evaluation framework that utilizes Chinese high school entrance examination (GAOKAO) questions as a dataset to evaluate the language understanding and logical reasoning abilities of large language models. GAOKAO-bench是一个以中国高考题目为数据集，旨在提供和人类对齐的，直观，高效地测评大模型语言理解能力、逻辑推理能力的测评框架", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:500", "languages": ["Chinese"], "modality": null, "name": "GAOKAO-Bench", "openness": "unknown", "publisher": null, "released": "2023-05-21", "released_reference": {"basis": "paper_first_version", "note": "First version of the GAOKAO benchmark paper, not a later paper revision.", "source_key": "opencompass:500", "source_url": "https://arxiv.org/abs/2305.12474"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-500-gaokao-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-Bench", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "ACL 2024", "多模态模型", "VLM", "知识储备", "Knowledge", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GAOKAO-MM is a multimodal benchmark based on the Chinese College Entrance Examination (GAOKAO), comprising of 8 subjects and 12 types of images, such as diagrams, function graphs, maps and photos. GAOKAO-MM 是一个基于中国高考的多模态基准，包含 8 个科目和 12 种图像类型，例如图表、函数图、地图和照片。GAOKAO-MM 源自本土中文语境，并对模型的能力设置了人类水平的要求，包括感知、理解、知识和推理。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1083", "languages": ["Chinese"], "modality": null, "name": "GAOKAO-MM", "openness": "unknown", "publisher": "OpenMOSS", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1083-gaokao-mm", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-MM", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "物理保真", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GAUGE: A Physical‑Realism Evaluation Benchmark Leveraging real‑world replicated experimental data as ground‑truth references, it delivers systematic evaluation for world models and physics engines across rigid bodies, ropes, fabrics, and 3D soft bodies. 物理真实度评测基准GAUGE，以真实重复实验数据为对照标准，为世界模型与物理引擎提供跨刚体、绳索、织物和三维软体的系统化评测。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:2574", "languages": [], "modality": null, "name": "GAUGE", "openness": "unknown", "publisher": "上海人工智能实验室", "released": "2026-08-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2574-gauge", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAUGE", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GDP.pdf is a knowledge-work vision benchmark that evaluates models on economically valuable professional tasks presented as visual documents (PDFs), testing document-based reasoning, chart and table interpretation, and problem solving without tools.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gdp-pdf:claude-fable-5", "reported_at": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gdp-pdf", "languages": [], "modality": "multimodal", "name": "GDP.pdf", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.6, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.816, "raw_min": 0.227, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gdp-pdf:claude-sonnet-5", "reported_date": "2026-06-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500"}, "unit": null}, "slug": "llm-stats-gdp-pdf", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500", "unit": null}, {"aliases": ["GDP.pdf"], "categories": ["multimodal"], "collected_at": null, "description": "Vision-based knowledge work over 100 PDFs across 10 domains, reported no-tools. Published by Surge AI with a public dataset (huggingface.co/datasets/surgeai/GDP.pdf), harness (github.com/surge-ai/gdp-pdf) and paper (arXiv:2607.11192). The name resembles GDPval but the two are separate instruments and must not be merged.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:gdp_pdf", "languages": [], "modality": null, "name": "GDP.pdf", "openness": "unknown", "publisher": null, "released": "2026-04-14", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:gdp_pdf", "source_url": "https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "gdp_pdf", "source": "model_reports", "source_url": "https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "finance", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GDPval is an OpenAI benchmark evaluating AI models on economically valuable, real-world knowledge-work tasks spanning many professional occupations and industries.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gdpval:nemotron-3-ultra-550b-a55b", "reported_at": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gdpval", "languages": [], "modality": "text", "name": "GDPval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.9, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.879, "raw_min": 0.467, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gdpval:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500"}, "unit": null}, "slug": "llm-stats-gdpval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500", "unit": null}, {"aliases": ["GDPval", "GDPval-AA", "GDPval-AA v2"], "categories": ["professional"], "collected_at": null, "description": "Graded by expert human comparison against real deliverables, and the \"-AA\" variants are run by Artificial Analysis rather than the vendor, so an Elo here is not comparable to a vendor-run pass rate.", "evidence_summary": {"document_count": 7, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000gdpval\u0000deepseek_v4_model_card\u0000gdpval_aa\u0000think max, Elo\u0000DeepSeek-V4-Pro", "reported_at": "2026-04-22", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"}, "first_score_reported_at": "2026-04-22", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000gdpval\u0000deepseek_v4_model_card\u0000gdpval_aa\u0000think max, Elo\u0000DeepSeek-V4-Pro", "reported_at": "2026-04-22", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:gdpval", "languages": [], "modality": null, "name": "GDPval", "openness": "unknown", "publisher": null, "released": "2025-09-25", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:gdpval", "source_url": "https://openai.com/index/gdpval/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 1861.0, "display_multiplier": 1, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 1861.0, "raw_min": 1395.0, "source_reference": {"obs_id": "curated\u0000gdpval\u0000anthropic_claude_opus_5_system_card\u0000gdpval_aa_v2\u0000ELO rating, max effort, agentic loop with shell access and web browsing\u0000Claude Opus 5", "observation_id": "curated\u0000gdpval\u0000anthropic_claude_opus_5_system_card\u0000gdpval_aa_v2\u0000ELO rating, max effort, agentic loop with shell access and web browsing\u0000Claude Opus 5", "reported_at": "2026-07-24", "reported_date": "2026-07-24", "source_id": "anthropic_claude_opus_5_system_card", "source_url": "https://www.anthropic.com/news/claude-opus-5"}, "unit": "elo"}, "slug": "gdpval", "source": "model_reports", "source_url": "https://openai.com/index/gdpval/", "unit": "elo"}, {"aliases": [], "categories": ["legal", "reasoning", "finance", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GDPval-AA evaluates AI agents on economically valuable professional knowledge-work tasks and reports performance as an Elo score.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gdpval-aa:inkling-small", "reported_at": "2026-07-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gdpval-aa", "languages": [], "modality": "text", "name": "GDPval-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 1769.0, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 1769.0, "raw_min": 1269.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gdpval-aa:glm-5.3", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500"}, "unit": null}, "slug": "llm-stats-gdpval-aa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "agentic"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic real-world work tasks", "evidence_summary": {"document_count": 1, "model_count": 213, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:gdpval-aa-v2:raw_elo:6a7c0e25-1dcb-4b15-8495-a8536a9da051", "reported_at": "2023-03-14", "source_url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:gdpval-aa-v2:raw_elo", "languages": [], "modality": null, "name": "GDPval-AA v2 (Elo)", "openness": "unknown", "publisher": null, "released": "2026-06-15", "released_reference": {"basis": "release_announcement", "note": "The publisher launches GDPval-AA v2 with human-anchored Elo, new judges and longer agent turns. The original GDPval dataset and GDPval-AA v1 have different release dates.", "source_key": "artificial-analysis:gdpval-aa-v2:raw_elo", "source_url": "https://artificialanalysis.ai/articles/artificial-analysis-intelligence-index-v4-1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 213, "score_direction": "higher_is_better", "score_summary": {"display_max": 1834.67, "display_multiplier": 1, "model_count": 213, "model_count_basis": "source_model_id", "numeric_count": 213, "raw_max": 1834.67, "raw_min": -122.53, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:gdpval-aa-v2:raw_elo:b8fc61f7-5e9a-49e6-8547-6ac56db24627", "reported_date": "2026-07-24", "source_url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}, "unit": null}, "slug": "artificial-analysis-gdpval-aa-v2-raw-elo", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/gdpval-aa", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "agentic"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic real-world work tasks", "evidence_summary": {"document_count": 1, "model_count": 213, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:gdpval-aa-v2:normalized_score:6a7c0e25-1dcb-4b15-8495-a8536a9da051", "reported_at": "2023-03-14", "source_url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:gdpval-aa-v2:normalized_score", "languages": [], "modality": null, "name": "GDPval-AA v2 (normalized score)", "openness": "unknown", "publisher": null, "released": "2026-06-15", "released_reference": {"basis": "release_announcement", "note": "The publisher launches GDPval-AA v2; this record is its normalized display score, kept separate from its raw-Elo source record.", "source_key": "artificial-analysis:gdpval-aa-v2:normalized_score", "source_url": "https://artificialanalysis.ai/articles/artificial-analysis-intelligence-index-v4-1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 213, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.7335, "display_multiplier": 100, "model_count": 213, "model_count_basis": "source_model_id", "numeric_count": 213, "raw_max": 0.667335, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:gdpval-aa-v2:normalized_score:b8fc61f7-5e9a-49e6-8547-6ac56db24627", "reported_date": "2026-07-24", "source_url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}, "unit": null}, "slug": "artificial-analysis-gdpval-aa-v2-normalized-score", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/gdpval-aa", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "finance", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GDPval-MM is the multimodal variant of the GDPval benchmark, evaluating AI model performance on real-world economically valuable tasks that require processing and generating multimodal content including documents, slides, diagrams, spreadsheets, images, and other professional deliverables across diverse industries.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gdpval-mm:minimax-m2.5", "reported_at": "2026-02-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gdpval-mm", "languages": [], "modality": "multimodal", "name": "GDPval-MM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.89999999999999, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.849, "raw_min": 0.59, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gdpval-mm:gpt-5.5", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500"}, "unit": null}, "slug": "llm-stats-gdpval-mm", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "finance", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GDPval-Rubrics evaluates AI model performance on economically valuable knowledge work tasks drawn from the public GDPval dataset. It uses pointwise scoring based on public rubrics, with the environment aligned to the GDPval-AA scaffolding.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gdpval-rubrics:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gdpval-rubrics", "languages": [], "modality": "text", "name": "GDPval-Rubrics", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.78, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.7478, "raw_min": 0.7478, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gdpval-rubrics:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500"}, "unit": null}, "slug": "llm-stats-gdpval-rubrics", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "其他", "Other", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GenAI-Bench is a benchmark designed to benchmark MLLMs’s ability in judging the quality of AI generative contents, containing over 40,000 human ratings to evaluate the performance of MLLMs on aligning with human preferences. GenAI-Bench用于衡量MLLM判断AI生成内容质量的能力，包含超过40000个人工评分用以评估大模型与人类偏好的一致性。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1376", "languages": [], "modality": "multimodal", "name": "GenAI-Bench", "openness": "open", "publisher": "Carnegie Mellon University", "released": "2024-06-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1376-genai-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GenAI-Bench", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GeneBench is an evaluation focused on multi-stage scientific data analysis in genetics and quantitative biology. Tasks require reasoning about ambiguous or noisy data with minimal supervisory guidance, addressing realistic obstacles such as hidden confounders or QC failures, and correctly implementing and interpreting modern statistical methods.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:genebench:gpt-5.5", "reported_at": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:genebench", "languages": [], "modality": "text", "name": "GeneBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 33.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.332, "raw_min": 0.25, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:genebench:gpt-5.5-pro", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500"}, "unit": null}, "slug": "llm-stats-genebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GeneBench-Pro is a research-level benchmark of 129 multi-stage computational-biology problems spanning genomics, quantitative biology, and translational biomedicine. Each problem gives the agent a messy dataset, brief context, and a target estimand, and requires navigating dependent inferential decision points to reach a verifiable answer.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:genebench-pro:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:genebench-pro", "languages": [], "modality": "text", "name": "GeneBench-Pro", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 28.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.287, "raw_min": 0.108, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:genebench-pro:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-genebench-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "General Bench is a set of universal evaluation benchmarks for multimodal large models, covering language, image, video, audio, and 3D five modalities, with a total of 145 skills, over 700 tasks, and 325800 samples. General-Bench 是一套面向多模态大模型的通用评测基准，涵盖语言、图像、视频、音频和 3D 五大模态，共计 145 项技能、700 余个任务，包含 325 800 条样本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2081", "languages": ["English", "Chinese"], "modality": "multimodal", "name": "General-Bench", "openness": "unknown", "publisher": "NUS , NTU , ZJU , KAUST , PKU , HFUT , UR , WHU , NJU , SJTU , Skywork AI", "released": "2025-05-07", "released_reference": {"basis": "paper_first_version", "note": "First version introducing General-Level and General-Bench.", "source_key": "opencompass:2081", "source_url": "https://arxiv.org/abs/2505.04620"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2081-general-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/General-Bench", "unit": null}, {"aliases": [], "categories": ["audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A dataset for tempo estimation in electronic dance music containing 664 2-minute audio previews from Beatport, annotated from user corrections for evaluating automatic tempo estimation algorithms.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:giantsteps-tempo:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:giantsteps-tempo", "languages": [], "modality": "audio", "name": "GiantSteps Tempo", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.88, "raw_min": 0.88, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:giantsteps-tempo:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500"}, "unit": null}, "slug": "llm-stats-giantsteps-tempo", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "智能体", "Agent", "任务执行", "Task Execution", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GitGoodBench Lite is a subset of 900 samples for evaluating the performance of AI agents in resolving git tasks (see Supported Scenarios). GitGoodBench Lite是一个包含900个样本的子集，用于评估AI智能体在解决git任务方面的性能（参见支持的场景）。数据集中的样本在编程语言Python、Java和Kotlin以及样本类型合并冲突解决和文件提交语法之间均匀分布。因此，该数据集包含每种样本类型和编程语言各150个样本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1865", "languages": ["English"], "modality": null, "name": "GitGoodBench", "openness": "open", "publisher": "JetBrains", "released": "2025-05-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1865-gitgoodbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GitGoodBench", "unit": null}, {"aliases": [], "categories": ["physics", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Global PIQA is a multilingual commonsense reasoning benchmark that evaluates physical interaction knowledge across 100 languages and cultures. It tests AI systems' understanding of physical world knowledge in diverse cultural contexts through multiple choice questions about everyday situations requiring physical commonsense.", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:global-piqa:gemini-3-pro-preview", "reported_at": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:global-piqa", "languages": [], "modality": "text", "name": "Global PIQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.4, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.934, "raw_min": 0.594, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:global-piqa:gemini-3-pro-preview", "reported_date": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500"}, "unit": null}, "slug": "llm-stats-global-piqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive multilingual benchmark covering 42 languages that addresses cultural and linguistic biases in evaluation, with improved translation quality and culturally sensitive question subsets.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:global-mmlu:gemma-3n-e2b-it-litert-preview", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:global-mmlu", "languages": [], "modality": "text", "name": "Global-MMLU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.6, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.836, "raw_min": 0.551, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:global-mmlu:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500"}, "unit": null}, "slug": "llm-stats-global-mmlu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A lightweight version of Global MMLU benchmark that evaluates language models across multiple languages while addressing cultural and linguistic biases in multilingual evaluation.", "evidence_summary": {"document_count": 1, "model_count": 15, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:global-mmlu-lite:gemini-2.0-flash-lite", "reported_at": "2025-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:global-mmlu-lite", "languages": [], "modality": "text", "name": "Global-MMLU-Lite", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 15, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.2, "display_multiplier": 100, "model_count": 15, "model_count_basis": "source_model_id", "numeric_count": 15, "raw_max": 0.892, "raw_min": 0.342, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:global-mmlu-lite:gemini-2.5-pro-preview-06-05", "reported_date": "2025-06-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500"}, "unit": null}, "slug": "llm-stats-global-mmlu-lite", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "其他", "Other", "医疗", "科学智能", "AI for Science", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GMAI-MMBench is the most comprehensive and structured benchmark developed to evaluate Large Vision-Language Models (LVLMs) in general medical artificial intelligence (GMAI) applications. It addresses the limitations of existing benchmarks that typically focus on narrow domains and lack perceptual di GMAI-MMBench 是目前最全面、结构化最完善的通用医学人工智能（GMAI）基准数据集，专为评估大规模视觉语言模型（LVLMs）在医疗领域中的表现而设计。针对现有医学基准通常局限于特定学科、感知粒度单一的问题，GMAI-MMBench 从285个真实医学数据集构建而成，涵盖39种医学影像模态、18个临床任务、18个医学科室，以及4种不同的感知粒度。所有任务采用视觉问答（VQA）形式组织，具备良好的交互性和通用性。其独特的词汇树结构允许用户针对具体研究需求灵活定制评估路径，支持多样化的评测场景。我们对50个主流LVLMs进行了系统测试，发现即使是目前最先进的 GPT-4o，其准确率也仅为5", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1927", "languages": [], "modality": "multimodal", "name": "GMAIMMBench", "openness": "unknown", "publisher": "上海人工智能实验室", "released": "2024-08-31", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1927-gmaimmbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GMAIMMBench", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Gorilla enables LLMs to use tools by invoking APIs. Given a natural language query, Gorilla comes up with the semantically- and syntactically- correct API to invoke. Gorilla 使大语言模型能够通过调用 API 使用工具。针对自然语言查询，Gorilla 能够生成语义和语法上正确的 API 调用。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1128", "languages": [], "modality": null, "name": "Gorilla", "openness": "unknown", "publisher": "Microsoft Research", "released": "2023-05-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1128-gorilla", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Gorilla", "unit": null}, {"aliases": [], "categories": ["reasoning", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "APIBench, a comprehensive dataset of over 11,000 instruction-API pairs from HuggingFace, TorchHub, and TensorHub APIs for evaluating language models' ability to generate accurate API calls.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gorilla-benchmark-api-bench:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gorilla-benchmark-api-bench", "languages": [], "modality": "text", "name": "Gorilla Benchmark API Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 35.3, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.353, "raw_min": 0.082, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gorilla-benchmark-api-bench:llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-gorilla-benchmark-api-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "summarization"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A long document summarization dataset consisting of reports from government research agencies including Congressional Research Service and U.S. Government Accountability Office, with significantly longer documents and summaries than other datasets.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:govreport:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:govreport", "languages": [], "modality": "text", "name": "GovReport", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 26.400000000000002, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.264, "raw_min": 0.259, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:govreport:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500"}, "unit": null}, "slug": "llm-stats-govreport", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500", "unit": null}, {"aliases": [], "categories": ["physics", "reasoning", "general", "biology", "chemistry"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. Questions are Google-proof and extremely difficult, with PhD experts reaching 65% accuracy.", "evidence_summary": {"document_count": 1, "model_count": 239, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gpqa:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:gpqa", "languages": [], "modality": "text", "name": "GPQA", "openness": "restricted", "publisher": "Anthropic", "released": "2023-11-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 239, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.6, "display_multiplier": 100, "model_count": 239, "model_count_basis": "source_model_id", "numeric_count": 239, "raw_max": 0.946, "raw_min": 0.119, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gpqa:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500"}, "unit": null}, "slug": "llm-stats-gpqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GPQA, a challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. GPQA 包含 448 道由生物学、物理学和化学领域专家撰写的多项选择题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1135", "languages": [], "modality": null, "name": "GPQA", "openness": "restricted", "publisher": "Anthropic", "released": "2023-11-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1135-gpqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GPQA", "unit": null}, {"aliases": ["GPQA main", "GPQA full"], "categories": ["science"], "collected_at": null, "description": "The full split, distinct from the Diamond subset most frontier cards report.", "evidence_summary": {"document_count": 3, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000gpqa\u0000qwen3_5_model_card\u0000gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000gpqa\u0000qwen3_5_model_card\u0000gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:gpqa", "languages": [], "modality": null, "name": "GPQA (full)", "openness": "unknown", "publisher": null, "released": "2023-11-20", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:gpqa", "source_url": "https://arxiv.org/abs/2311.12022"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 88.4, "raw_min": 88.4, "source_reference": {"obs_id": "curated\u0000gpqa\u0000qwen3_5_model_card\u0000gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000gpqa\u0000qwen3_5_model_card\u0000gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "gpqa", "source": "model_reports", "source_url": "https://arxiv.org/abs/2311.12022", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "general", "healthcare", "biology"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Biology subset of GPQA, containing challenging multiple-choice questions written by domain experts in biology. These Google-proof questions require graduate-level knowledge and reasoning.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gpqa-biology:o1-2024-12-17", "reported_at": "2024-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gpqa-biology", "languages": [], "modality": "text", "name": "GPQA Biology", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 69.19999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.692, "raw_min": 0.692, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gpqa-biology:o1-2024-12-17", "reported_date": "2024-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500"}, "unit": null}, "slug": "llm-stats-gpqa-biology", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "chemistry"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Chemistry subset of GPQA, containing challenging multiple-choice questions written by domain experts in chemistry. These Google-proof questions require graduate-level knowledge and reasoning.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gpqa-chemistry:o1-2024-12-17", "reported_at": "2024-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gpqa-chemistry", "languages": [], "modality": "text", "name": "GPQA Chemistry", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.647, "raw_min": 0.647, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gpqa-chemistry:o1-2024-12-17", "reported_date": "2024-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500"}, "unit": null}, "slug": "llm-stats-gpqa-chemistry", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "reasoning", "science"], "collected_at": "2026-08-25T10:41:06Z", "description": "Scientific reasoning", "evidence_summary": {"document_count": 1, "model_count": 586, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:gpqa-diamond:037dec2f-51e8-4127-a1f1-85155dae7a1d", "reported_at": "2022-11-30", "source_url": "https://artificialanalysis.ai/evaluations/gpqa-diamond"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:gpqa-diamond", "languages": [], "modality": null, "name": "GPQA Diamond", "openness": "unknown", "publisher": null, "released": "2023-11-20", "released_reference": {"basis": "paper_first_version", "note": "First version of the GPQA paper defining its Diamond partition. This remains before the requested 2024 cohort.", "source_key": "artificial-analysis:gpqa-diamond", "source_url": "https://arxiv.org/abs/2311.12022"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 586, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.94949494949499, "display_multiplier": 100, "model_count": 586, "model_count_basis": "source_model_id", "numeric_count": 586, "raw_max": 0.94949494949495, "raw_min": 0.097979797979798, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:gpqa-diamond:c8adc5cf-fd5a-407b-af51-dc3bede3e49c", "reported_date": "2026-08-12", "source_url": "https://artificialanalysis.ai/evaluations/gpqa-diamond"}, "unit": null}, "slug": "artificial-analysis-gpqa-diamond", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/gpqa-diamond", "unit": null}, {"aliases": ["GPQA", "GPQA-Diamond", "GPQA Diamond"], "categories": ["science"], "collected_at": null, "description": "198 questions in the Diamond split, so run-to-run variance is wide and a single reported number hides it. Approaching saturation at the 2026 frontier, where reported scores cluster in the high 80s and 90s.", "evidence_summary": {"document_count": 27, "model_count": 19, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000gpqa_diamond\u0000deepseek_v3_report\u0000gpqa_diamond\u0000Pass@1, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000gpqa_diamond\u0000deepseek_v3_report\u0000gpqa_diamond\u0000Pass@1, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:gpqa_diamond", "languages": [], "modality": null, "name": "GPQA Diamond", "openness": "unknown", "publisher": null, "released": "2023-11-20", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:gpqa_diamond", "source_url": "https://arxiv.org/abs/2311.12022"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 21, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.3, "display_multiplier": 1, "model_count": 19, "model_count_basis": "source_model_id", "numeric_count": 21, "raw_max": 94.3, "raw_min": 59.1, "source_reference": {"obs_id": "curated\u0000gpqa_diamond\u0000google_gemini_3_1_pro_model_card\u0000gpqa_diamond\u0000Thinking (High), No tools\u0000Gemini 3.1 Pro", "observation_id": "curated\u0000gpqa_diamond\u0000google_gemini_3_1_pro_model_card\u0000gpqa_diamond\u0000Thinking (High), No tools\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "reported_date": "2026-02-19", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "unit": "percent"}, "slug": "gpqa_diamond", "source": "model_reports", "source_url": "https://arxiv.org/abs/2311.12022", "unit": "percent"}, {"aliases": [], "categories": ["physics", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Physics subset of GPQA, containing challenging multiple-choice questions written by domain experts in physics. These Google-proof questions require graduate-level knowledge and reasoning.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gpqa-physics:o1-2024-12-17", "reported_at": "2024-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gpqa-physics", "languages": [], "modality": "text", "name": "GPQA Physics", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.80000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.928, "raw_min": 0.928, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gpqa-physics:o1-2024-12-17", "reported_date": "2024-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500"}, "unit": null}, "slug": "llm-stats-gpqa-physics", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GPT-ImgEval, quantitatively and qualitatively diagnoses GPT-4o's performance across three critical dimensions: (1) generation quality, (2) editing proficiency, and (3) world knowledge-informed semantic synthesis. GPT-ImgEval，从三个关键维度对GPT-4o的性能进行定量和定性诊断：（1）生成质量，（2）编辑能力，以及（3）基于世界知识的语义合成能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1740", "languages": [], "modality": "multimodal", "name": "GPT-ImgEval", "openness": "open", "publisher": "Peking University, Sun Yat-sen University, etc.", "released": "2025-04-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1740-gpt-imgeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GPT-ImgEval", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GrailQA is a new large-scale, high-quality dataset for question answering on knowledge bases (KBQA) on Freebase with 64,331 questions annotated with both answers and corresponding logical forms in different syntax (i.e., SPARQL, S-expression, etc.). GrailQA 是一个大规模高质量数据集，用于知识库问答，包含 64,331 个问题，并附有答案和不同语法的相应逻辑形式。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1098", "languages": [], "modality": null, "name": "GrailQA", "openness": "unknown", "publisher": "The Ohio State University", "released": "2021-02-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1098-grailqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GrailQA", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GraphWalks is a synthetic multi-hop long-context reasoning benchmark in which a model is given an edge-list representation of a graph and must traverse it to find neighboring nodes (via breadth-first search) or parent nodes for a given start node. Performance is reported as F1 of the model-predicted answer set versus the ground truth.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:graphwalks:mimo-v2.5", "reported_at": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:graphwalks", "languages": [], "modality": "text", "name": "GraphWalks", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.9, "raw_min": 0.62, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:graphwalks:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500"}, "unit": null}, "slug": "llm-stats-graphwalks", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "spatial_reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "GraphWalks BFS variant evaluated on 1M-token contexts.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:graphwalks-bfs-1m:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:graphwalks-bfs-1m", "languages": [], "modality": "text", "name": "Graphwalks BFS 1M", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.10000000000001, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.771, "raw_min": 0.512, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:graphwalks-bfs-1m:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500"}, "unit": null}, "slug": "llm-stats-graphwalks-bfs-1m", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "spatial_reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A graph reasoning benchmark that evaluates language models' ability to perform breadth-first search (BFS) operations on graphs with context length under 128k tokens, returning nodes reachable at specified depths.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:graphwalks-bfs-<128k:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:graphwalks-bfs-<128k", "languages": [], "modality": "text", "name": "Graphwalks BFS <128k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.0, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.94, "raw_min": 0.25, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:graphwalks-bfs-<128k:gpt-5.2-2025-12-11", "reported_date": "2025-12-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500"}, "unit": null}, "slug": "llm-stats-graphwalks-bfs-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "spatial_reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A graph reasoning benchmark that evaluates language models' ability to perform breadth-first search (BFS) operations on graphs with context length over 128k tokens, testing long-context reasoning capabilities.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:graphwalks-bfs->128k:gpt-4.1-2025-04-14", "reported_at": "2025-04-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:graphwalks-bfs->128k", "languages": [], "modality": "text", "name": "Graphwalks BFS >128k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.7, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.907, "raw_min": 0.029, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:graphwalks-bfs->128k:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500"}, "unit": null}, "slug": "llm-stats-graphwalks-bfs-128k-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "spatial_reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A graph reasoning benchmark that evaluates language models' ability to find parent nodes in graphs with context length under 128k tokens, requiring understanding of graph structure and edge relationships.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:graphwalks-parents-<128k:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:graphwalks-parents-<128k", "languages": [], "modality": "text", "name": "Graphwalks parents <128k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.8, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.898, "raw_min": 0.094, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:graphwalks-parents-<128k:gpt-5.4", "reported_date": "2026-03-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500"}, "unit": null}, "slug": "llm-stats-graphwalks-parents-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "spatial_reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A graph reasoning benchmark that evaluates language models' ability to find parent nodes in graphs with context length over 128k tokens, testing long-context reasoning and graph structure understanding.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:graphwalks-parents->128k:gpt-4.1-2025-04-14", "reported_at": "2025-04-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:graphwalks-parents->128k", "languages": [], "modality": "text", "name": "Graphwalks parents >128k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.39999999999999, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.954, "raw_min": 0.056, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:graphwalks-parents->128k:claude-opus-4-6", "reported_date": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500"}, "unit": null}, "slug": "llm-stats-graphwalks-parents-128k-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GroundingSuite is designed to test the localization capabilities of multimodal models. It created 3,720 pixel-level data entries based on COCO Unlabeled images. This dataset covers four dimensions: Stuff Class Object, Multi-Object, Part-Level Object, and Single Object. GroundingSuite 用来测试多模态模型的定位能力。它通过半自动标注和人工筛选在COCO Unlabel的图片基础上创建了3720条pixel-level的数据，覆盖Stuff Class Object, Multi Object, Part Level Object, Single Object四个维度，是一个全面评测多模态模型定位能力的测试集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2027", "languages": [], "modality": "multimodal", "name": "GroundingSuite", "openness": "open", "publisher": "Huazhong University of Science and Technology", "released": "2025-07-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2027-groundingsuite", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GroundingSuite", "unit": null}, {"aliases": [], "categories": ["multimodal", "grounding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A subset of GroundUI-18K for UI grounding evaluation, where models must predict action coordinates on screenshots based on single-step instructions across web, desktop, and mobile platforms.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:groundui-1k:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:groundui-1k", "languages": [], "modality": "multimodal", "name": "GroundUI-1K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.39999999999999, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.814, "raw_min": 0.802, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:groundui-1k:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500"}, "unit": null}, "slug": "llm-stats-groundui-1k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Grade School Math 8K with Chain-of-Thought prompting, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gsm-8k-(cot):llama-3.1-70b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gsm-8k-(cot)", "languages": [], "modality": "text", "name": "GSM-8K (CoT)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.1, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.951, "raw_min": 0.845, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gsm-8k-(cot):llama-3.1-70b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500"}, "unit": null}, "slug": "llm-stats-gsm-8k-cot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "NeurIPS 2024", "大语言模型", "LLM", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GSM1k is meant for evaluating LLM's math reasoning ability. It mirrors the style and complexity of the established GSM8k benchmark while consider the problem of data-leaking and overfitting GSM1k可用于评估LLM的数学推理能力；它与GSM8k保持了风格和复杂性的一致，同时考虑了数据泄露和过拟合的问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1272", "languages": [], "modality": null, "name": "GSM1k", "openness": "unknown", "publisher": null, "released": "2024-05-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1272-gsm1k", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GSM1k", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Grade School Math 8K, a dataset of 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.", "evidence_summary": {"document_count": 1, "model_count": 48, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gsm8k:claude-3-opus-20240229", "reported_at": "2024-02-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:gsm8k", "languages": [], "modality": "text", "name": "GSM8k", "openness": "unknown", "publisher": null, "released": "2021-04-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 48, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.6, "display_multiplier": 100, "model_count": 48, "model_count_basis": "source_model_id", "numeric_count": 48, "raw_max": 0.996, "raw_min": 0.252, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gsm8k:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500"}, "unit": null}, "slug": "llm-stats-gsm8k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500", "unit": null}, {"aliases": ["GSM8K"], "categories": ["math"], "collected_at": null, "description": "Saturated. A legacy baseline for small open models.", "evidence_summary": {"document_count": 7, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000gsm8k\u0000google_gemini_1_5_report\u0000gsm8k\u000011-shot\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "first_score_reported_at": "2024-03-08", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000gsm8k\u0000google_gemini_1_5_report\u0000gsm8k\u000011-shot\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:gsm8k", "languages": [], "modality": null, "name": "GSM8K", "openness": "unknown", "publisher": null, "released": "2021-10-27", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:gsm8k", "source_url": "https://github.com/openai/grade-school-math"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.8, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 90.8, "raw_min": 90.8, "source_reference": {"obs_id": "curated\u0000gsm8k\u0000google_gemini_1_5_report\u0000gsm8k\u000011-shot\u0000Gemini 1.5 Pro", "observation_id": "curated\u0000gsm8k\u0000google_gemini_1_5_report\u0000gsm8k\u000011-shot\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "reported_date": "2024-03-08", "source_id": "google_gemini_1_5_report", "source_url": "https://arxiv.org/abs/2403.05530"}, "unit": "percent"}, "slug": "gsm8k", "source": "model_reports", "source_url": "https://github.com/openai/grade-school-math", "unit": "percent"}, {"aliases": [], "categories": ["数学", "Math", "大语言模型", "LLM", "数理能力", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GSM8K is a dataset of 8,500 high quality linguistically diverse grade school math word problems created by human problem writers. The dataset is segmented into 7,500 training problems and 1,000 test problems. These problems take between 2 and 8 steps to solve, and solutions primarily involve performing a sequence of elementary calculations using basic arithmetic operations (+ − × ÷) to reach the final answer. GSM8K 是一个包含 8,500 个高质量、语言多样化的小学数学单词问题的数据集，由人类问题编写者创建。该数据集分为 7,500 个训练问题和 1,000 个测试问题。这些问题的解题步骤在 2 到 8 步之间，解题过程主要涉及使用基本算术运算（+ - × ÷）进行一连串的基本计算，从而得出最终答案。一个聪明的初中生应该能够解决每一个问题。它可用于多步数学推理。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:535", "languages": [], "modality": null, "name": "GSM8K", "openness": "unknown", "publisher": null, "released": "2021-04-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-535-gsm8k", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Grade School Math 8K adapted for chat format evaluation, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:gsm8k-chat:llama-3.1-nemotron-70b-instruct", "reported_at": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:gsm8k-chat", "languages": [], "modality": "text", "name": "GSM8K Chat", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.88, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.8188, "raw_min": 0.8188, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:gsm8k-chat:llama-3.1-nemotron-70b-instruct", "reported_date": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500"}, "unit": null}, "slug": "llm-stats-gsm8k-chat", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "数学", "Math", "visual mathematical reasoning", "gsm8k", "multi-image math benchmark", "多模态模型", "VLM", "逻辑推理", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GSM8K-V is a multi-image, purely visual mathematical reasoning benchmark built by rendering GSM8K textual problems into visual scenes. It exposes substantial gaps in current vision-language models’ visual reasoning despite their near-saturated performance on text-based GSM8K. GSM8K-V 是一个多图视觉数学推理基准，通过将 GSM8K 的文本题系统性转换为多场景图像构建而成。它揭示了当前 VLMs 在视觉数学推理上的表现与其在文本数学推理上的表现之间存在显著差距。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2233", "languages": [], "modality": "multimodal", "name": "GSM8K-V", "openness": "open", "publisher": "浙江大学", "released": "2025-09-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2233-gsm8k-v", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K-V", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "NeurIPS 2024", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "GTA is meant for LLM's tool-use evaluations under real-world scenarios, including 229 real-world tasks and executable tool chains. GTA用于评估LLM调用工具解决实际问题的能力，由229个真实任务和可执行工具链组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1328", "languages": [], "modality": null, "name": "GTA", "openness": "open", "publisher": "Shanghai AI Laboratory", "released": "2024-07-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1328-gta", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GTA", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "强化学习", "智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Gym4ReaL is a benchmark suite designed to evaluate reinforcement learning algorithms in real-world scenarios, addressing challenges such as non-stationarity, partial observability, and large state-action spaces. Gym4ReaL 是一个用于评估强化学习算法在真实世界场景中表现的基准套件，涵盖非平稳性、部分可观测性和大状态-动作空间等挑战。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2035", "languages": [], "modality": null, "name": "Gym4ReaL", "openness": "unknown", "publisher": "Politecnico di Milano", "released": "2025-06-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2035-gym4real", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Gym4ReaL", "unit": null}, {"aliases": [], "categories": ["reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive benchmark designed to evaluate image-context reasoning in large visual-language models (LVLMs) by challenging models with 346 images and 1,129 carefully crafted questions to assess language hallucination and visual illusion", "evidence_summary": {"document_count": 1, "model_count": 18, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hallusion-bench:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hallusion-bench", "languages": [], "modality": "multimodal", "name": "Hallusion Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 18, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.0, "display_multiplier": 100, "model_count": 18, "model_count_basis": "source_model_id", "numeric_count": 18, "raw_max": 0.7, "raw_min": 0.472, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hallusion-bench:qwen3.5-27b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-hallusion-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "图像理解", "Image Understanding", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HallusionBench is a comprehensive benchmark designed for the evaluation of image-context reasoning, comprising 346 images paired with 1129 questions. HallusionBench是一个专为评估图像上下文推理而设计的综合基准测试，包括346张图像和1129个问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1355", "languages": [], "modality": "multimodal", "name": "HallusionBench", "openness": "unknown", "publisher": "University of Maryland", "released": "2023-10-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1355-hallusionbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HallusionBench", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HaluEval evaluates the performance of LLMs in recognizing hallucination. It includes 5,000 general user queries with ChatGPT responses and 30,000 task-specific examples from three tasks, i.e., question answering, knowledge-grounded dialogue, and text summarization. HaluEval用于评估大语言模型识别幻觉的能力，包含 5,000 条普通用户查询及 ChatGPT 的回答，以及来自三个任务的 30,000 个特定任务示例，即问答、基于知识的对话和文本摘要。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1121", "languages": [], "modality": null, "name": "HaluEval", "openness": "unknown", "publisher": "Beijing Key Laboratory of Big Data Management and Analysis Methods", "released": "2023-10-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1121-halueval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HaluEval", "unit": null}, {"aliases": ["Harbor-Index", "Harbor Index"], "categories": ["coding_agent"], "collected_at": null, "description": "Compact high-signal frontier-agent evaluation; the environment and harness define the score.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000harbor_index\u0000tencent_hy4_preview\u0000harbor_index\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000harbor_index\u0000tencent_hy4_preview\u0000harbor_index\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:harbor_index", "languages": [], "modality": null, "name": "Harbor-Index", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 39.6, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 39.6, "raw_min": 39.6, "source_reference": {"obs_id": "curated\u0000harbor_index\u0000tencent_hy4_preview\u0000harbor_index\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000harbor_index\u0000tencent_hy4_preview\u0000harbor_index\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "harbor_index", "source": "model_reports", "source_url": "https://github.com/harbor-framework/harbor-index", "unit": "percent"}, {"aliases": [], "categories": ["推理", "Reasoning", "代码", "Code", "数学", "Math", "Strong Reasoning", "大语言模型", "LLM", "逻辑推理", "代码工程", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HARDMath2 is a benchmark for applied mathematics created by students in a graduate class at Harvard University, featuring 211 original problems covering core topics such as boundary-layer analysis, WKB methods, asymptotic solutions of nonlinear partial differential equations, and the asymptotics. HARDMath2是由哈佛大学研究生课程的学生创建的一项应用数学基准测试，包含211道原创问题，涵盖边界层分析、WKB方法、非线性偏微分方程的渐近解以及振荡积分的渐近性等核心主题。该基准通过一种创新的协作方式构建，学生不仅设计并改进符合课程大纲的高难度问题，还对解决方案进行同行验证，同时测试不同模型的表现。最终，LLM生成的解答会与学生的答案以及数值真值进行自动对比，以评估模型的准确性和能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1842", "languages": [], "modality": null, "name": "HARDMath2", "openness": "unknown", "publisher": "Harvard University", "released": "2025-05-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1842-hardmath2", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HARDMath2", "unit": null}, {"aliases": [], "categories": ["knowledge", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Harvey LAB (Vals) is a professional legal-work evaluation of AI systems on complex law-firm style tasks, reported by Vals.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:harvey-lab:grok-4.6", "reported_at": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:harvey-lab", "languages": [], "modality": "text", "name": "Harvey LAB (Vals)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 15.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.158, "raw_min": 0.158, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:harvey-lab:grok-4.6", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500"}, "unit": null}, "slug": "llm-stats-harvey-lab", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500", "unit": null}, {"aliases": [], "categories": ["agentic", "legal"], "collected_at": "2026-08-25T10:41:06Z", "description": "Legal agentic work, criterion pass rate", "evidence_summary": {"document_count": 1, "model_count": 36, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:harvey-lab-aa:f0083258-8646-45b8-8082-7aaf6c2ea82a", "reported_at": "2025-08-05", "source_url": "https://artificialanalysis.ai/evaluations/harvey-lab-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:harvey-lab-aa", "languages": [], "modality": null, "name": "Harvey LAB-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 36, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.6446663771892, "display_multiplier": 100, "model_count": 36, "model_count_basis": "source_model_id", "numeric_count": 36, "raw_max": 0.946446663771892, "raw_min": 0.138949196699957, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:harvey-lab-aa:f7d2fc3e-1f7b-405f-818c-07952a4af78f", "reported_date": "2026-07-16", "source_url": "https://artificialanalysis.ai/evaluations/harvey-lab-aa"}, "unit": null}, "slug": "artificial-analysis-harvey-lab-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/harvey-lab-aa", "unit": null}, {"aliases": [], "categories": ["knowledge", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Harvey LAB-AA evaluates model performance on complex legal workflows.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:harvey-lab-aa:gemini-3.7-flash", "reported_at": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:harvey-lab-aa", "languages": [], "modality": "text", "name": "Harvey LAB-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.907, "raw_min": 0.907, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:harvey-lab-aa:gemini-3.7-flash", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500"}, "unit": null}, "slug": "llm-stats-harvey-lab-aa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500", "unit": null}, {"aliases": [], "categories": ["healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "An open-source benchmark for measuring performance and safety of large language models in healthcare, consisting of 5,000 multi-turn conversations evaluated by 262 physicians using 48,562 unique rubric criteria across health contexts and behavioral dimensions", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:healthbench:gpt-oss-120b", "reported_at": "2025-08-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:healthbench", "languages": [], "modality": "text", "name": "HealthBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.199999999999996, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.602, "raw_min": 0.425, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:healthbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500"}, "unit": null}, "slug": "llm-stats-healthbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500", "unit": null}, {"aliases": ["HealthBench", "HealthBench Hard", "HealthBench Consensus"], "categories": ["health"], "collected_at": null, "description": "Physician-written rubrics scored by an LLM grader, and the length-adjusted variant is what recent cards report. Published by OpenAI.", "evidence_summary": {"document_count": 2, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:healthbench", "languages": [], "modality": null, "name": "HealthBench", "openness": "unknown", "publisher": null, "released": "2025-05-12", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:healthbench", "source_url": "https://openai.com/index/healthbench/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "healthbench", "source": "model_reports", "source_url": "https://openai.com/index/healthbench/", "unit": null}, {"aliases": [], "categories": ["healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "HealthBench Consensus is a HealthBench subset focused on questions where physician-created rubric criteria have especially high agreement, measuring healthcare performance and safety on consensus-evaluable conversations.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:healthbench-consensus:gpt-5.5-instant", "reported_at": "2026-05-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:healthbench-consensus", "languages": [], "modality": "text", "name": "HealthBench Consensus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.5, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.955, "raw_min": 0.947, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:healthbench-consensus:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500"}, "unit": null}, "slug": "llm-stats-healthbench-consensus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500", "unit": null}, {"aliases": [], "categories": ["healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A challenging variation of HealthBench that evaluates large language models' performance and safety in healthcare through 5,000 multi-turn conversations with particularly rigorous evaluation criteria validated by 262 physicians from 60 countries", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:healthbench-hard:gpt-oss-120b", "reported_at": "2025-08-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:healthbench-hard", "languages": [], "modality": "text", "name": "HealthBench Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 42.8, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.428, "raw_min": 0.016, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:healthbench-hard:muse-spark", "reported_date": "2026-04-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-healthbench-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500", "unit": null}, {"aliases": [], "categories": ["healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "HealthBench Professional evaluates model capability and safety for clinician use cases using real clinician-style chats and physician-authored grading rubrics.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:healthbench-professional:gpt-5.5-instant", "reported_at": "2026-05-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:healthbench-professional", "languages": [], "modality": "text", "name": "HealthBench Professional", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.0, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.66, "raw_min": 0.35, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:healthbench-professional:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500"}, "unit": null}, "slug": "llm-stats-healthbench-professional", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500", "unit": null}, {"aliases": ["HealthBench Professional"], "categories": ["health"], "collected_at": null, "description": "A distinct, harder split from the HealthBench variants recorded separately in this registry, so its scores are not comparable to them. Published as an OpenAI paper (525 examples, 3 clinician use cases) with a public dataset; released 2026-04-22 per PDF metadata. Graded by an LLM against physician-written rubrics.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:healthbench_professional", "languages": [], "modality": null, "name": "HealthBench Professional", "openness": "unknown", "publisher": null, "released": "2026-04-22", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:healthbench_professional", "source_url": "https://cdn.openai.com/dd128428-0184-4e25-b155-3a7686c7d744/HealthBench-Professional.pdf"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "healthbench_professional", "source": "model_reports", "source_url": "https://cdn.openai.com/dd128428-0184-4e25-b155-3a7686c7d744/HealthBench-Professional.pdf", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A challenging commonsense natural language inference dataset that uses Adversarial Filtering to create questions trivial for humans (>95% accuracy) but difficult for state-of-the-art models, requiring completion of sentence endings based on physical situations and everyday activities", "evidence_summary": {"document_count": 1, "model_count": 27, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hellaswag:gpt-4-0613", "reported_at": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:hellaswag", "languages": [], "modality": "text", "name": "HellaSwag", "openness": "restricted", "publisher": null, "released": "2019-05-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 27, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.39999999999999, "display_multiplier": 100, "model_count": 27, "model_count_basis": "source_model_id", "numeric_count": 27, "raw_max": 0.954, "raw_min": 0.33, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hellaswag:claude-3-opus-20240229", "reported_date": "2024-02-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500"}, "unit": null}, "slug": "llm-stats-hellaswag", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HellaSwag is a challenge dataset for evaluating commonsense natural language inference, which is specially hard for state-of-the-art models, though its questions are trivial for humans (>95% accuracy). It consists of 70k multiple choice questions, each with a scenario and four possible endings, which requires to select the most reasonable ending. These questions come from two domains:activitynet and wikihow, involving video and text scenarios respectively. The correct answers of these questions are the real sentences for the next event, while the incorrect answers are adversarially generated and human verified, so as to fool machines but not humans. HellaSwag 是一个用于评估常识性自然语言推理的数据集，HellaSwag的问题对于最先进的模型来说是特别困难的，尽管它的问题对于人类来说非常轻松就能回答的（> 95% 的准确率）。它由7万多道多项选择题组成，每道题都有一个场景和四种可能的答案，需要选择最合理的答案。这些问题来自两个领域：activitynet和wikihow，分别涉及视频和文本场景。这些问题的正确答案是下一个事件的真实句子，而错误答案是通过对抗技术生成的并经过人类验证，这些答案可以欺骗机器但不能欺骗人类。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:531", "languages": [], "modality": null, "name": "HellaSwag", "openness": "restricted", "publisher": null, "released": "2019-05-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-531-hellaswag", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HellaSwag", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "大语言模型", "LLM", "语言生成", "Generation", "长上下文", "Long Context", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HelloBench is a hierarchical long text generation benchmark to evaluate LLMs' performance in generating long text.  Based on Bloom's Taxonomy, HelloBench categorizes long text generation tasks into five subtasks: open-ended QA, summarization, chat, text completion, and heuristic text generation. HelloBench为长文本生成基准，这是一个全面的、开放式的基准，用于评估LLM在生成长文本方面的性能。基于Bloom的分类法，HelloBench将长文本生成任务分为五个子任务：开放式QA、摘要、聊天、文本完成和启发式文本生成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1097", "languages": [], "modality": null, "name": "HelloBench", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-09-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1097-hellobench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HelloBench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "RAG", "大语言模型", "LLM", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HERB comprises 39,190 artifacts—including documents, meeting transcripts, Slack messages, GitHub content, and URLs—simulating business workflows across product planning, development, and support stages, with noisy, multi-hop QA tasks featuring both answerable and unanswerable queries. HERB基准包含39,190个企业文档、会议记录、Slack消息、GitHub内容和网页链接，模拟产品规划、开发与支持等业务流程，生成包含噪声的多跳问答任务，涵盖可回答与不可回答查询。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:2031", "languages": [], "modality": null, "name": "HERB", "openness": "unknown", "publisher": "Salesforce AI Research", "released": "2025-06-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2031-herb", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HERB", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Google DeepMind's internal mathematical reasoning benchmark that introduces novel problems not encountered during model training to evaluate true mathematical reasoning capabilities rather than memorization", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hiddenmath:gemini-1.5-flash-8b", "reported_at": "2024-03-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hiddenmath", "languages": [], "modality": "text", "name": "HiddenMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.0, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.63, "raw_min": 0.158, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hiddenmath:gemini-2.0-flash", "reported_date": "2024-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500"}, "unit": null}, "slug": "llm-stats-hiddenmath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "science", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "HiPhO is a high-school physics olympiad benchmark evaluating multimodal reasoning over physics problems that include diagrams and figures.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hipho:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hipho", "languages": [], "modality": "multimodal", "name": "HiPhO", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.841, "raw_min": 0.841, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hipho:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500"}, "unit": null}, "slug": "llm-stats-hipho", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "knowledge"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "HLE-Verified evaluates multidisciplinary expert reasoning on a verified subset of Humanity's Last Exam.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hle-verified:gemini-3.7-flash", "reported_at": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hle-verified", "languages": [], "modality": "text", "name": "HLE-Verified", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 53.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.536, "raw_min": 0.536, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hle-verified:gemini-3.7-flash", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500"}, "unit": null}, "slug": "llm-stats-hle-verified", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500", "unit": null}, {"aliases": ["HMMT", "HMMT Feb 25", "HMMT Nov 25", "HMMT 2026 Feb"], "categories": ["math"], "collected_at": null, "description": "A dated competition, not a fixed set: each sitting is a different instrument and contamination rises once problems are published.", "evidence_summary": {"document_count": 6, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000hmmt\u0000zai_glm_5_model_card\u0000hmmt\u0000HMMT Nov. 2025, temp 1.0, top_p 0.95, GPT-5.2 (medium) judge\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000hmmt\u0000zai_glm_5_model_card\u0000hmmt\u0000HMMT Nov. 2025, temp 1.0, top_p 0.95, GPT-5.2 (medium) judge\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:hmmt", "languages": [], "modality": null, "name": "HMMT", "openness": "unknown", "publisher": null, "released": "2025-02-15", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:hmmt", "source_url": "https://www.hmmt.org/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.9, "display_multiplier": 1, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 96.9, "raw_min": 92.7, "source_reference": {"obs_id": "curated\u0000hmmt\u0000zai_glm_5_model_card\u0000hmmt\u0000HMMT Nov. 2025, temp 1.0, top_p 0.95, GPT-5.2 (medium) judge\u0000GLM-5", "observation_id": "curated\u0000hmmt\u0000zai_glm_5_model_card\u0000hmmt\u0000HMMT Nov. 2025, temp 1.0, top_p 0.95, GPT-5.2 (medium) judge\u0000GLM-5", "reported_at": "2026-02-11", "reported_date": "2026-02-11", "source_id": "zai_glm_5_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "unit": "percent"}, "slug": "hmmt", "source": "model_reports", "source_url": "https://www.hmmt.org/", "unit": "percent"}, {"aliases": [], "categories": ["math"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds", "evidence_summary": {"document_count": 1, "model_count": 33, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hmmt-2025:deepseek-v3.1", "reported_at": "2025-01-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:hmmt-2025", "languages": [], "modality": "text", "name": "HMMT 2025", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 33, "score_direction": "higher_is_better", "score_summary": {"display_max": 100.0, "display_multiplier": 100, "model_count": 33, "model_count_basis": "source_model_id", "numeric_count": 33, "raw_max": 1.0, "raw_min": 0.289, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hmmt-2025:gpt-5.2-pro-2025-12-11", "reported_date": "2025-12-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500"}, "unit": null}, "slug": "llm-stats-hmmt-2025", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "HMMT February 2026 is a math competition benchmark based on problems from the Harvard-MIT Mathematics Tournament, testing advanced mathematical problem-solving and reasoning.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hmmt-feb-26:qwen3.6-plus", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hmmt-feb-26", "languages": [], "modality": "text", "name": "HMMT Feb 26", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.1, "display_multiplier": 100, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 0.971, "raw_min": 0.826, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hmmt-feb-26:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500"}, "unit": null}, "slug": "llm-stats-hmmt-feb-26", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500", "unit": null}, {"aliases": [], "categories": ["math"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds", "evidence_summary": {"document_count": 1, "model_count": 25, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hmmt25:grok-4", "reported_at": "2025-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hmmt25", "languages": [], "modality": "text", "name": "HMMT25", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "needs_review", "score_count": 25, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.7, "display_multiplier": 100, "model_count": 25, "model_count_basis": "source_model_id", "numeric_count": 25, "raw_max": 0.967, "raw_min": 0.307, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hmmt25:grok-4-heavy", "reported_date": "2025-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500"}, "unit": null}, "slug": "llm-stats-hmmt25", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "长文本", "Long-Context", "大语言模型", "LLM", "逻辑推理", "长上下文", "Long Context", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HoloBench is a benchmark designed to evaluate the ability of long-context language models (LCLMs) to perform holistic reasoning over extended text contexts. HoloBench 是一个用于评估长上下文语言模型（LCLMs）在扩展文本上下文中进行整体推理能力的基准测试。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1581", "languages": [], "modality": null, "name": "HoloBench", "openness": "unknown", "publisher": "Megagon Labs", "released": "2024-10-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1581-holobench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HoloBench", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "HorizonMath is an extremely difficult frontier mathematics benchmark designed to test the limits of mathematical reasoning on research-level and competition-beyond problems.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:horizonmath:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:horizonmath", "languages": [], "modality": "text", "name": "HorizonMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 7.1, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.071, "raw_min": 0.02, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:horizonmath:hy3", "reported_date": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500"}, "unit": null}, "slug": "llm-stats-horizonmath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500", "unit": null}, {"aliases": [], "categories": ["math"], "collected_at": null, "description": "Unsolved research-level mathematics, reported at pass@4, so figures sit far below conventional accuracy scales.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000horizonmath\u0000tencent_hy4_preview\u0000horizonmath\u0000pass@4\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000horizonmath\u0000tencent_hy4_preview\u0000horizonmath\u0000pass@4\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:horizonmath", "languages": [], "modality": null, "name": "HorizonMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 8.8, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 8.8, "raw_min": 8.8, "source_reference": {"obs_id": "curated\u0000horizonmath\u0000tencent_hy4_preview\u0000horizonmath\u0000pass@4\u0000Hy4 preview", "observation_id": "curated\u0000horizonmath\u0000tencent_hy4_preview\u0000horizonmath\u0000pass@4\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "horizonmath", "source": "model_reports", "source_url": "https://github.com/ewang26/HorizonMath", "unit": "percent"}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "大语言模型", "LLM", "知识储备", "Knowledge", "逻辑推理", "Reasoning", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HOTPOTQA is a dataset with 113k Wikipedia-based question-answer pairs. HotpotQA 用于评估大语言模型的推理能力，包含 113,000 个基于维基百科的问题和答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1108", "languages": [], "modality": null, "name": "HotpotQA", "openness": "open", "publisher": "Google AI", "released": "2018-09-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1108-hotpotqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HotpotQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "HR-Bench (4k) evaluates image understanding on high-resolution visual inputs with a 4k setting.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hr-bench-4k:mimo-v2.5", "reported_at": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hr-bench-4k", "languages": [], "modality": "multimodal", "name": "HR-Bench (4k)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.885, "raw_min": 0.885, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hr-bench-4k:mimo-v2.5", "reported_date": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500"}, "unit": null}, "slug": "llm-stats-hr-bench-4k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "其他", "Other", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HtFLlib is a benchmark for heterogeneous federated learning that examines how 40 vision, NLP and sensor models and 10 algorithms collaborate under non-IID data. HtFLlib 是一个面向异构联邦学习算法的综合评测基准，旨在衡量不同模型架构在非 IID 数据环境中的协同学习能力。评测对象覆盖图像、文本与传感信号三类模型，总计 40 个架构及 10 种代表性方法。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1992", "languages": [], "modality": null, "name": "HtFLlib", "openness": "unknown", "publisher": "Shanghai Jiao Tong University , Beihang University , Chongqing University , etc.", "released": "2025-06-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1992-htfllib", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HtFLlib", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "evidence_summary": {"document_count": 1, "model_count": 66, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humaneval:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:humaneval", "languages": [], "modality": "text", "name": "HumanEval", "openness": "unknown", "publisher": null, "released": "2021-07-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 66, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.12, "display_multiplier": 100, "model_count": 66, "model_count_basis": "source_model_id", "numeric_count": 66, "raw_max": 0.9512, "raw_min": 0.348, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humaneval:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500"}, "unit": null}, "slug": "llm-stats-humaneval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500", "unit": null}, {"aliases": ["HumanEval", "HumanEval+"], "categories": ["coding"], "collected_at": null, "description": "Saturated. Retained because open-weight cards still report it.", "evidence_summary": {"document_count": 9, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000humaneval\u0000google_gemini_1_5_report\u0000humaneval\u00000-shot, chat preamble\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "first_score_reported_at": "2024-03-08", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000humaneval\u0000google_gemini_1_5_report\u0000humaneval\u00000-shot, chat preamble\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:humaneval", "languages": [], "modality": null, "name": "HumanEval", "openness": "unknown", "publisher": null, "released": "2021-07-07", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:humaneval", "source_url": "https://github.com/openai/human-eval"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.7, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 92.7, "raw_min": 84.1, "source_reference": {"obs_id": "curated\u0000humaneval\u0000qwen2_5_coder_report\u0000humaneval\u0000HumanEval base split\u0000Qwen2.5-Coder-32B-Instruct", "observation_id": "curated\u0000humaneval\u0000qwen2_5_coder_report\u0000humaneval\u0000HumanEval base split\u0000Qwen2.5-Coder-32B-Instruct", "reported_at": "2024-09-18", "reported_date": "2024-09-18", "source_id": "qwen2_5_coder_report", "source_url": "https://arxiv.org/abs/2409.12186"}, "unit": "percent"}, "slug": "humaneval", "source": "model_reports", "source_url": "https://github.com/openai/human-eval", "unit": "percent"}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The benchmark consists of around 1,000 crowd-sourced Python programming problems, designed to be solvable by entry level programmers, covering programming fundamentals, standard library functionality, and so on. Each problem consists of a task description, code solution and 3 automated test cases. 这是 \"Evaluating Large Language Models Trained on Code\" 论文中描述的 HumanEval 问题解决数据集的评估工具包。它用于测量从文档脚本合成程序的功能正确性。它由 164 个原始编程问题组成，评估语言理解能力、算法和简单数学，其中一些问题与简单的软件面试题类似。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:537", "languages": [], "modality": null, "name": "HumanEval", "openness": "unknown", "publisher": null, "released": "2021-07-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-537-humaneval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Enhanced version of HumanEval that extends the original test cases by 80x using EvalPlus framework for rigorous evaluation of LLM-synthesized code functional correctness, detecting previously undetected wrong code", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humaneval-plus:mistral-small-3.2-24b-instruct-2506", "reported_at": "2025-06-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humaneval-plus", "languages": [], "modality": "text", "name": "HumanEval Plus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.929, "raw_min": 0.929, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humaneval-plus:mistral-small-3.2-24b-instruct-2506", "reported_date": "2025-06-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500"}, "unit": null}, "slug": "llm-stats-humaneval-plus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Enhanced version of HumanEval that extends the original test cases by 80x using EvalPlus framework for rigorous evaluation of LLM-synthesized code functional correctness, detecting previously undetected wrong code", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humaneval+:qwen-2.5-14b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humaneval+", "languages": [], "modality": "text", "name": "HumanEval+", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.9, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.929, "raw_min": 0.25, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humaneval+:phi-4-reasoning", "reported_date": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500"}, "unit": null}, "slug": "llm-stats-humaneval-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humaneval-average:codestral-22b", "reported_at": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humaneval-average", "languages": [], "modality": "text", "name": "HumanEval-Average", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.615, "raw_min": 0.615, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humaneval-average:codestral-22b", "reported_date": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500"}, "unit": null}, "slug": "llm-stats-humaneval-average", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humaneval-er:kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humaneval-er", "languages": [], "modality": "text", "name": "HumanEval-ER", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.10000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.811, "raw_min": 0.811, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humaneval-er:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500"}, "unit": null}, "slug": "llm-stats-humaneval-er", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multilingual variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humaneval-mul:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humaneval-mul", "languages": [], "modality": "text", "name": "HumanEval-Mul", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.826, "raw_min": 0.738, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humaneval-mul:deepseek-v3", "reported_date": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500"}, "unit": null}, "slug": "llm-stats-humaneval-mul", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HumanEval-X is a benchmark for evaluating the multilingual ability of code generative models. It consists of 820 high-quality human-crafted data samples (each with test cases) in Python, C++, Java, JavaScript, and Go, and can be used for various tasks, such as code generation and translation. HumanEval-X 是一个用于评估代码生成模型的多语言能力的基准测试。它包含了820个高质量的人工制作的数据样本（每个都有测试案例），包括Python、C++、Java、JavaScript和Go语言，可用于各种任务，如代码生成和翻译。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:543", "languages": ["Multilingual"], "modality": null, "name": "HumanEval-X", "openness": "open", "publisher": null, "released": "2023-03-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-543-humaneval-x", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval-X", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Average evaluation of HumanEval Fill-in-the-Middle benchmark variants (single-line, multi-line, random-span) for assessing code infilling capabilities of language models", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humanevalfim-average:codestral-22b", "reported_at": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humanevalfim-average", "languages": [], "modality": "text", "name": "HumanEvalFIM-Average", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.60000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.916, "raw_min": 0.916, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humanevalfim-average:codestral-22b", "reported_date": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500"}, "unit": null}, "slug": "llm-stats-humanevalfim-average", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Humanity's Last Exam (HLE) is a multi-modal academic benchmark with 2,500 questions across mathematics, humanities, and natural sciences, designed to test LLM capabilities at the frontier of human knowledge with unambiguous, verifiable solutions", "evidence_summary": {"document_count": 1, "model_count": 99, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humanity's-last-exam:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:humanity's-last-exam", "languages": [], "modality": "multimodal", "name": "Humanity's Last Exam", "openness": "restricted", "publisher": "Center for AI Safety", "released": "2025-01-24", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 99, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.7, "display_multiplier": 100, "model_count": 99, "model_count_basis": "source_model_id", "numeric_count": 99, "raw_max": 0.647, "raw_min": 0.037, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humanity's-last-exam:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500"}, "unit": null}, "slug": "llm-stats-humanity-s-last-exam", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500", "unit": null}, {"aliases": ["HLE", "Humanity's Last Exam", "HLE-Full", "HLE-Verified"], "categories": ["reasoning"], "collected_at": null, "description": "Tool access and test-time compute budget move this score more than model capability does, so two reported figures are rarely comparable. Cards increasingly report no-tools and with-tools figures as separate numbers.", "evidence_summary": {"document_count": 20, "model_count": 16, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000hle\u0000google_gemini_2_5_report\u0000hle\u0000no tools\u0000Gemini 2.5 Pro", "reported_at": "2025-06-17", "source_url": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf"}, "first_score_reported_at": "2025-06-17", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000hle\u0000google_gemini_2_5_report\u0000hle\u0000no tools\u0000Gemini 2.5 Pro", "reported_at": "2025-06-17", "source_url": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:hle", "languages": [], "modality": null, "name": "Humanity's Last Exam", "openness": "unknown", "publisher": null, "released": "2025-01-23", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:hle", "source_url": "https://lastexam.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 27, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.0, "display_multiplier": 1, "model_count": 16, "model_count_basis": "source_model_id", "numeric_count": 27, "raw_max": 60.0, "raw_min": 7.7, "source_reference": {"obs_id": "curated\u0000hle\u0000deepseek_v4_pro_0813_model_card\u0000hle\u0000max reasoning effort, temp 1.0, top_p 0.95, with tools\u0000DeepSeek-V4-Pro-0813", "observation_id": "curated\u0000hle\u0000deepseek_v4_pro_0813_model_card\u0000hle\u0000max reasoning effort, temp 1.0, top_p 0.95, with tools\u0000DeepSeek-V4-Pro-0813", "reported_at": "2026-08-13", "reported_date": "2026-08-13", "source_id": "deepseek_v4_pro_0813_model_card", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"}, "unit": "percent"}, "slug": "hle", "source": "model_reports", "source_url": "https://lastexam.ai/", "unit": "percent"}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Text-only Humanity's Last Exam variant evaluated without tool use.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humanity's-last-exam-(no-tools,-text-only):hy3", "reported_at": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humanity's-last-exam-(no-tools,-text-only)", "languages": [], "modality": "text", "name": "Humanity's Last Exam (no tools, text-only)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 47.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.47, "raw_min": 0.427, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humanity's-last-exam-(no-tools,-text-only):hy3", "reported_date": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500"}, "unit": null}, "slug": "llm-stats-humanity-s-last-exam-no-tools-text-only", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Text-only Humanity's Last Exam variant evaluated with tool use enabled.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:humanity's-last-exam-(with-tools,-text-only):hy3", "reported_at": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:humanity's-last-exam-(with-tools,-text-only)", "languages": [], "modality": "text", "name": "Humanity's Last Exam (with tools, text-only)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.6, "raw_min": 0.532, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:humanity's-last-exam-(with-tools,-text-only):deepseek-v4-pro-0813", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500"}, "unit": null}, "slug": "llm-stats-humanity-s-last-exam-with-tools-text-only", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "reasoning", "knowledge"], "collected_at": "2026-08-25T10:41:06Z", "description": "Reasoning & knowledge", "evidence_summary": {"document_count": 1, "model_count": 577, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:humanitys-last-exam:83cb898e-05d9-4e4b-9de3-2d305014d923", "reported_at": "2023-03-14", "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:humanitys-last-exam", "languages": [], "modality": null, "name": "Humanity’s Last Exam", "openness": "unknown", "publisher": null, "released": "2025-01-23", "released_reference": {"basis": "release_announcement", "note": "Scale and CAIS announce the public HLE benchmark and results on January 23, before the January 24 arXiv submission. GPT-4 model releases in 2023 do not date HLE.", "source_key": "artificial-analysis:humanitys-last-exam", "source_url": "https://scale.com/blog/humanitys-last-exam-results"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 577, "score_direction": "higher_is_better", "score_summary": {"display_max": 55.468025949953706, "display_multiplier": 100, "model_count": 577, "model_count_basis": "source_model_id", "numeric_count": 577, "raw_max": 0.554680259499537, "raw_min": 0.0106580166821131, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:humanitys-last-exam:cd55210d-358e-4df1-ba9c-9acb5f186cc9", "reported_date": "2026-06-09", "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam"}, "unit": null}, "slug": "artificial-analysis-humanitys-last-exam", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam", "unit": null}, {"aliases": [], "categories": ["spatial_reasoning", "3d", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Hypersim evaluates 3D grounding and depth understanding in synthetic indoor scenes.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:hypersim:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:hypersim", "languages": [], "modality": "image", "name": "Hypersim", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.131, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.131, "raw_min": 0.11, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:hypersim:qwen3.5-35b-a3b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500"}, "unit": null}, "slug": "llm-stats-hypersim", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "综合能力", "Comprehensive Capability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HypoBench, a novel benchmark designed to evaluate LLMs and hypothesis generation methods across multiple aspects, including practical utility, generalizability, and hypothesis discovery rate. HypoBench includes 7 real-world tasks and 5 synthetic tasks with 194 distinct datasets. HypoBench，这是一种新颖的基准，旨在从多个方面评估 LLM 和假设生成方法，包括实用性、泛化性和假设发现率。HypoBench 包括 7 个真实任务和 5 个合成任务，具有 194 个不同的数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1782", "languages": [], "modality": null, "name": "HypoBench", "openness": "unknown", "publisher": "University of Chicago， University of Toronto", "released": "2025-04-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1782-hypobench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HypoBench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "语言生成", "Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "HypoEval, Hypothesis-guided Evaluation framework, which first uses a small corpus of human evaluations to generate more detailed rubrics for human judgments and then incorporates a checklist-like approach to combine LLM's assigned scores on each decomposed dimension to acquire overall scores. HypoEval，即假设指导的评估框架，该框架首先使用一小部分人工评估来生成更详细的人类判断量规，然后采用类似清单的方法，将 LLM 在每个分解维度上的分配分数结合起来，以获得总分。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1786", "languages": [], "modality": null, "name": "HypoEval", "openness": "unknown", "publisher": "University of Chicago", "released": "2025-04-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1786-hypoeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HypoEval", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "NeurIPS 2024", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IaC-Eval is meant for quantitatively evaluating the capabilities of LLMs in cloud IaC code generation, containing 458 questions ranging from simple to difficult across various cloud services. IaC-Eval用于定量评估LLM在云IaC代码生成中的功能，其中包含458个从易到难的问题，涵盖了各种云服务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1320", "languages": [], "modality": null, "name": "IaC-Eval", "openness": "open", "publisher": "University of Michigan", "released": "2024-09-26", "released_reference": {"basis": "paper_first_version", "note": "OpenReview lists September 26 as the introducing paper's public publication date; private submission and later modification dates are not used.", "source_key": "opencompass:1320", "source_url": "https://openreview.net/forum?id=7TCK0aBL1C"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1320-iac-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IaC-Eval", "unit": null}, {"aliases": [], "categories": ["structured_output", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Instruction-Following Evaluation (IFEval) benchmark for large language models, focusing on verifiable instructions with 25 types of instructions and around 500 prompts containing one or more verifiable constraints", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:if:mistral-small-3.2-24b-instruct-2506", "reported_at": "2025-06-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:if", "languages": [], "modality": "text", "name": "IF", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.78, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.8478, "raw_min": 0.72, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:if:mistral-small-3.2-24b-instruct-2506", "reported_date": "2025-06-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500"}, "unit": null}, "slug": "llm-stats-if", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500", "unit": null}, {"aliases": [], "categories": ["instruction-following"], "collected_at": "2026-08-25T10:41:06Z", "description": "Instruction following", "evidence_summary": {"document_count": 1, "model_count": 450, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:ifbench:6a7c0e25-1dcb-4b15-8495-a8536a9da051", "reported_at": "2023-03-14", "source_url": "https://artificialanalysis.ai/evaluations/ifbench"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:ifbench", "languages": [], "modality": null, "name": "IFBench", "openness": "unknown", "publisher": null, "released": "2025-07-03", "released_reference": {"basis": "paper_first_version", "note": "First version of Generalizing Verifiable Instruction Following introduces IFBench and its 58 new constraint types.", "source_key": "artificial-analysis:ifbench", "source_url": "https://arxiv.org/abs/2507.02833"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 450, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.3333333333333, "display_multiplier": 100, "model_count": 450, "model_count_basis": "source_model_id", "numeric_count": 450, "raw_max": 0.833333333333333, "raw_min": 0.12108843537415, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:ifbench:5e8b0d98-a3b4-42b5-93d8-ecb748788754", "reported_date": "2026-04-30", "source_url": "https://artificialanalysis.ai/evaluations/ifbench"}, "unit": null}, "slug": "artificial-analysis-ifbench", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/ifbench", "unit": null}, {"aliases": [], "categories": ["instruction_following", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Instruction Following Benchmark evaluating model's ability to follow complex instructions", "evidence_summary": {"document_count": 1, "model_count": 34, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ifbench:hermes-3-70b", "reported_at": "2024-08-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:ifbench", "languages": [], "modality": "text", "name": "IFBench", "openness": "open", "publisher": "Allen Institute for AI", "released": "2025-07-03", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 34, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.8, "display_multiplier": 100, "model_count": 34, "model_count_basis": "source_model_id", "numeric_count": 34, "raw_max": 0.828, "raw_min": 0.21, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ifbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500"}, "unit": null}, "slug": "llm-stats-ifbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500", "unit": null}, {"aliases": ["IFBench"], "categories": ["instruction_following"], "collected_at": null, "description": "Verifiable constraints on unseen instruction types; does not measure answer quality.", "evidence_summary": {"document_count": 3, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000ifbench\u0000qwen3_5_model_card\u0000ifbench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000ifbench\u0000qwen3_5_model_card\u0000ifbench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:ifbench", "languages": [], "modality": null, "name": "IFBench", "openness": "unknown", "publisher": null, "released": "2025-07-03", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:ifbench", "source_url": "https://arxiv.org/abs/2507.02833"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.5, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 76.5, "raw_min": 76.5, "source_reference": {"obs_id": "curated\u0000ifbench\u0000qwen3_5_model_card\u0000ifbench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000ifbench\u0000qwen3_5_model_card\u0000ifbench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "ifbench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2507.02833", "unit": "percent"}, {"aliases": [], "categories": ["structured_output", "instruction_following", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Instruction-Following Evaluation (IFEval) benchmark for large language models, focusing on verifiable instructions with 25 types of instructions and around 500 prompts containing one or more verifiable constraints", "evidence_summary": {"document_count": 1, "model_count": 67, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ifeval:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:ifeval", "languages": [], "modality": "text", "name": "IFEval", "openness": "unknown", "publisher": "Google", "released": "2023-11-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 67, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.0, "display_multiplier": 100, "model_count": 67, "model_count_basis": "source_model_id", "numeric_count": 67, "raw_max": 0.95, "raw_min": 0.44, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ifeval:qwen3.5-27b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500"}, "unit": null}, "slug": "llm-stats-ifeval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500", "unit": null}, {"aliases": ["IFEval"], "categories": ["instruction_following"], "collected_at": null, "description": "Verifiable-constraint subset only; does not measure answer quality.", "evidence_summary": {"document_count": 9, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000ifeval\u0000deepseek_v3_report\u0000ifeval\u0000Prompt Strict\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000ifeval\u0000deepseek_v3_report\u0000ifeval\u0000Prompt Strict\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:ifeval", "languages": [], "modality": null, "name": "IFEval", "openness": "unknown", "publisher": null, "released": "2023-11-14", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:ifeval", "source_url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.6, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 92.6, "raw_min": 83.3, "source_reference": {"obs_id": "curated\u0000ifeval\u0000qwen3_5_model_card\u0000ifeval\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000ifeval\u0000qwen3_5_model_card\u0000ifeval\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "ifeval", "source": "model_reports", "source_url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval", "unit": "percent"}, {"aliases": [], "categories": ["指令跟随", "Instruct", "大语言模型", "LLM", "指令遵循", "Instruction Following", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IFEval is a straightforward and easy-to reproduce evaluation benchmark. It focuses on a set of “verifiable instructions” such as “write in more than 400 words” and “mention the keyword of AI at least 3 times”. IFEval 是一个简单且易于复现的评估基准。它关注一组“可验证的指令”，例如“写超过 400 个单词”和“至少提到关键词 AI 3 次”。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1136", "languages": [], "modality": null, "name": "IFEval", "openness": "unknown", "publisher": "Google", "released": "2023-11-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1136-ifeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IFEval", "unit": null}, {"aliases": [], "categories": ["指令跟随", "Instruct", "大语言模型", "LLM", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IFIR is the first comprehensive benchmark designed to evaluate instruction-following information retrieval (IR) in expert domains. IFIR includes 2,426 high-quality examples and covers eight subsets across four specialized domains: finance, law, healthcare, and science literature. IFIR，这是第一个旨在评估专家领域指令跟随信息检索（IR）的综合基准。IFIR 包含 2,426 个高质量示例，涵盖四个专业领域（金融、法律、医疗保健和科学文献）的八个子集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1617", "languages": [], "modality": null, "name": "IFIR", "openness": "unknown", "publisher": "UCAS, ZJU, etc.", "released": "2025-03-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1617-ifir", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IFIR", "unit": null}, {"aliases": [], "categories": ["multimodal", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Image2FloorPlan is an in-house benchmark evaluating multimodal models on generating structured floor plans and interactive frontends directly from images such as design mockups and room photos.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:image2floorplan:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:image2floorplan", "languages": [], "modality": "multimodal", "name": "Image2FloorPlan", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.48, "raw_min": 0.359, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:image2floorplan:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500"}, "unit": null}, "slug": "llm-stats-image2floorplan", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ImageMining evaluates multimodal models on extracting structured information from images using tool use, measuring ability to combine visual understanding with tool-based retrieval and analysis.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:imagemining:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:imagemining", "languages": [], "modality": "multimodal", "name": "ImageMining", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 30.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.307, "raw_min": 0.307, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:imagemining:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500"}, "unit": null}, "slug": "llm-stats-imagemining", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "NeurIPS 2024", "多模态模型", "VLM", "安全对齐", "Safety Alignment", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IMDL-BenCo offers a comprehensive IMDL benchmark and modular codebase. IMDL-BenCo提供了全面的IMDL基准测试和模块化代码库。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1274", "languages": [], "modality": "multimodal", "name": "IMDL-BenCo", "openness": "unknown", "publisher": "Sichuan University", "released": "2024-06-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1274-imdl-benco", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IMDL-BenCo", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "IMO 2025 evaluates models on the six problems from the 2025 International Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:imo-2025:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:imo-2025", "languages": [], "modality": "text", "name": "IMO 2025", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 35.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 35.0, "raw_min": 0.652, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:imo-2025:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500"}, "unit": null}, "slug": "llm-stats-imo-2025", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "IMO-AnswerBench is a benchmark for evaluating mathematical reasoning capabilities on International Mathematical Olympiad (IMO) problems, focusing on answer generation and verification.", "evidence_summary": {"document_count": 1, "model_count": 20, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:imo-answerbench:kimi-k2-thinking-0905", "reported_at": "2025-09-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:imo-answerbench", "languages": [], "modality": "text", "name": "IMO-AnswerBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "monorepo_subdir", "repo_resolution_status": "resolved", "score_count": 20, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.30000000000001, "display_multiplier": 100, "model_count": 20, "model_count_basis": "source_model_id", "numeric_count": 20, "raw_max": 0.923, "raw_min": 0.783, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:imo-answerbench:nemotron-3-ultra-550b-a55b", "reported_date": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500"}, "unit": null}, "slug": "llm-stats-imo-answerbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500", "unit": null}, {"aliases": ["IMOAnswerBench", "IMO AnswerBench"], "categories": ["math"], "collected_at": null, "description": "Final-answer grading on olympiad problems, so it does not check whether the proof reasoning was valid.", "evidence_summary": {"document_count": 6, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000imo_answer_bench\u0000zai_glm_5_model_card\u0000imo_answer_bench\u0000temp 1.0, top_p 0.95, GPT-5.2 (medium) judge\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000imo_answer_bench\u0000zai_glm_5_model_card\u0000imo_answer_bench\u0000temp 1.0, top_p 0.95, GPT-5.2 (medium) judge\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:imo_answer_bench", "languages": [], "modality": null, "name": "IMOAnswerBench", "openness": "unknown", "publisher": null, "released": "2025-09-18", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:imo_answer_bench", "source_url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.0, "display_multiplier": 1, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 91.0, "raw_min": 80.9, "source_reference": {"obs_id": "curated\u0000imo_answer_bench\u0000zai_glm_5_2_model_card\u0000imo_answer_bench\u0000temp 1.0, top_p 0.95, max gen 163840 tokens, GPT-5.5 (medium) judge\u0000GLM-5.2", "observation_id": "curated\u0000imo_answer_bench\u0000zai_glm_5_2_model_card\u0000imo_answer_bench\u0000temp 1.0, top_p 0.95, max gen 163840 tokens, GPT-5.5 (medium) judge\u0000GLM-5.2", "reported_at": "2026-06-16", "reported_date": "2026-06-16", "source_id": "zai_glm_5_2_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5.2"}, "unit": "percent"}, "slug": "imo_answer_bench", "source": "model_reports", "source_url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench", "unit": "percent"}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "IMOProof-Adv is an advanced benchmark of International Mathematical Olympiad-style proof problems requiring rigorous multi-step mathematical proofs.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:imoproof-adv:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:imoproof-adv", "languages": [], "modality": "text", "name": "IMOProof-Adv", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 54.300000000000004, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.543, "raw_min": 0.543, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:imoproof-adv:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500"}, "unit": null}, "slug": "llm-stats-imoproof-adv", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Include benchmark - specific documentation not found in official sources", "evidence_summary": {"document_count": 1, "model_count": 31, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:include:qwen3-235b-a22b", "reported_at": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:include", "languages": [], "modality": "text", "name": "Include", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 31, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.6, "display_multiplier": 100, "model_count": 31, "model_count_basis": "source_model_id", "numeric_count": 31, "raw_max": 0.876, "raw_min": 0.386, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:include:claude-opus-4-8", "reported_date": "2026-05-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500"}, "unit": null}, "slug": "llm-stats-include", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:community:07c9946d-dcf0-4977-a640-a6b1356b4f0b", "languages": [], "modality": null, "name": "independence-bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-community-07c9946d-dcf0-4977-a640-a6b1356b4f0b", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A07c9946d-dcf0-4977-a640-a6b1356b4f0b?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IndicMMLU-Pro provides a standardized evaluation framework to push the research boundaries in Indic language AI, facilitating the development of more accurate, efficient, and culturally sensitive models. IndicMMLU-Pro 提供了一个标准化的评估框架，以推动印度语系语言 AI 的研究边界，促进更准确、高效和具有文化敏感性的模型的发展。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1670", "languages": [], "modality": null, "name": "IndicMMLU-Pro", "openness": "restricted", "publisher": "Artificial Intelligence Institute, University of South Carolina, etc.", "released": "2025-01-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1670-indicmmlu-pro", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IndicMMLU-Pro", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "NeurIPS 2024", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "InfiBench is a large-scale freeform question-answering (QA) benchmark for code to our knowledge, comprising 234 carefully selected high-quality Stack Overflow questions that span across 15 programming languages. InfiBench用于评测LLM回答代码相关问题的能力，包括涵盖15种编程语言的234个精心挑选的高质量Stack Overflow问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1322", "languages": [], "modality": null, "name": "InfiBench", "openness": "open", "publisher": "Simon Fraser University", "released": "2024-03-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1322-infibench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InfiBench", "unit": null}, {"aliases": [], "categories": ["long_context"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "InfiniteBench English Multiple Choice variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:infinitebench-en.mc:llama-3.2-3b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:infinitebench-en.mc", "languages": [], "modality": "text", "name": "InfiniteBench/En.MC", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.633, "raw_min": 0.633, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:infinitebench-en.mc:llama-3.2-3b-instruct", "reported_date": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500"}, "unit": null}, "slug": "llm-stats-infinitebench-en-mc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "InfiniteBench English Question Answering variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:infinitebench-en.qa:llama-3.2-3b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:infinitebench-en.qa", "languages": [], "modality": "text", "name": "InfiniteBench/En.QA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 19.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.198, "raw_min": 0.198, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:infinitebench-en.qa:llama-3.2-3b-instruct", "reported_date": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500"}, "unit": null}, "slug": "llm-stats-infinitebench-en-qa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500", "unit": null}, {"aliases": [], "categories": ["指令跟随", "Instruct", "ACL 2024", "大语言模型", "LLM", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "InfoBench is a benchmark comprising 500 diverse instructions and 2,250 decomposed questions across multiple constraint categories. InfoBench 是一个指令追随评测基准，包含 500 条多样化的指令和 2,250 个分解问题，涵盖多个约束类别。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1085", "languages": [], "modality": null, "name": "InfoBench", "openness": "open", "publisher": "Tencent AI Lab", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1085-infobench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InfoBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "InfographicVQA dataset with 5,485 infographic images and over 30,000 questions requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:infographicsqa:llama-3.2-90b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:infographicsqa", "languages": [], "modality": "multimodal", "name": "InfographicsQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.568, "raw_min": 0.568, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:infographicsqa:llama-3.2-90b-instruct", "reported_date": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500"}, "unit": null}, "slug": "llm-stats-infographicsqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "InfoVQA dataset with 30,000 questions and 5,000 infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:infovqa:deepseek-vl2", "reported_at": "2024-12-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:infovqa", "languages": [], "modality": "multimodal", "name": "InfoVQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.39999999999999, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.834, "raw_min": 0.5, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:infovqa:qwen2.5-vl-32b", "reported_date": "2025-02-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500"}, "unit": null}, "slug": "llm-stats-infovqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "InfoVQA test set with infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:infovqatest:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:infovqatest", "languages": [], "modality": "multimodal", "name": "InfoVQAtest", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.60000000000001, "display_multiplier": 100, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 0.926, "raw_min": 0.803, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:infovqatest:kimi-k2.5", "reported_date": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500"}, "unit": null}, "slug": "llm-stats-infovqatest", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Instruction-based variant of HumanEval benchmark for evaluating large language models' code generation capabilities with functional correctness using pass@k metric on programming problems", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:instruct-humaneval:llama-3.1-nemotron-70b-instruct", "reported_at": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:instruct-humaneval", "languages": [], "modality": "text", "name": "Instruct HumanEval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.83999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.7384, "raw_min": 0.7384, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:instruct-humaneval:llama-3.1-nemotron-70b-instruct", "reported_date": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500"}, "unit": null}, "slug": "llm-stats-instruct-humaneval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "NAACL 2024", "大语言模型", "LLM", "语言理解", "Comprehension", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "InstruSum evaluates the task of instruction controllable text summarization, where the model input consists of both a source article and a natural language requirement for desired summary characteristics. InstruSum 是评测指令可控的文本摘要任务，模型输入包括源文章和对所需摘要特征的自然语言要求。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1149", "languages": [], "modality": null, "name": "InstruSum", "openness": "unknown", "publisher": "Allen Institute for AI", "released": "2024-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1149-instrusum", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InstruSum", "unit": null}, {"aliases": [], "categories": ["math", "spatial_reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Interpretable Geometry Problem Solver (Inter-GPS) with Geometry3K dataset of 3,002 geometry problems with dense annotation in formal language using theorem knowledge and symbolic reasoning", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:intergps:phi-3.5-vision-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:intergps", "languages": [], "modality": "text", "name": "InterGPS", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.486, "raw_min": 0.363, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:intergps:phi-4-multimodal-instruct", "reported_date": "2025-02-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500"}, "unit": null}, "slug": "llm-stats-intergps", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500", "unit": null}, {"aliases": [], "categories": ["structured_output", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Internal API instruction following (hard) benchmark - specific documentation not found in official sources", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:internal-api-instruction-following-(hard):gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:internal-api-instruction-following-(hard)", "languages": [], "modality": "text", "name": "Internal API instruction following (hard)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.0, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.64, "raw_min": 0.292, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:internal-api-instruction-following-(hard):gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500"}, "unit": null}, "slug": "llm-stats-internal-api-instruction-following-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The Internal Research Debugging Evaluation measures whether models can debug 41 real bugs from internal OpenAI research experiments (plus alignment-auditing tasks), where the original solutions took experienced researchers hours to days. Passing corresponds to providing assistance that would unblock the user, including partial root-cause explanations or fixes.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:internal-research-debugging-evaluation:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:internal-research-debugging-evaluation", "languages": [], "modality": "text", "name": "Internal Research Debugging Evaluation", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.30000000000001, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.683, "raw_min": 0.508, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:internal-research-debugging-evaluation:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500"}, "unit": null}, "slug": "llm-stats-internal-research-debugging-evaluation", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "video", "物理智能", "Embodied AI", "跨模态推理", "Cross-modal Reasoning", "空间理解", "Spatial Understanding", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "InternData-A1: A hybrid synthetic-real manipulation dataset integrating 5 heterogeneous robots, 15 skills, and 200+ scenes, emphasizing multi-robot collaboration under dynamic scenarios. InternData-A1：一个融合了 5 种异构机器人、15 项技能和 200+ 场景的混合合成-真实操作数据集，重点关注动态场景下的多机器人协作。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": false, "key": "opencompass:2151", "languages": [], "modality": "multimodal", "name": "InternData-A1", "openness": "restricted", "publisher": "InternRobotics", "released": "2025-07-26", "released_reference": {"basis": "dataset_published", "note": "The official dataset card explicitly states its release date.", "source_key": "opencompass:2151", "source_url": "https://huggingface.co/datasets/InternRobotics/InternData-A1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2151-interndata-a1", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InternData-A1", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "物理智能", "Embodied AI", "跨模态推理", "Cross-modal Reasoning", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "InternData-M1: A large-scale synthetic dataset for generalizable pick-and-place over 80K objects, with open-ended instructions covering object recognition, spatial and commonsense reasoning, and long-horizon tasks. InternData-M1：一个大规模合成数据集，用于可泛化的抓取与放置任务，涵盖 8 万+ 对象，配有开放式指令，涉及物体识别、空间与常识推理以及长时任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": false, "key": "opencompass:2153", "languages": [], "modality": "multimodal", "name": "InternData-M1", "openness": "restricted", "publisher": null, "released": "2025-07-26", "released_reference": {"basis": "dataset_published", "note": "The official dataset license and changelog give the same initial release date.", "source_key": "opencompass:2153", "source_url": "https://huggingface.co/datasets/InternRobotics/InternData-M1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2153-interndata-m1", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InternData-M1", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "物理智能", "Embodied AI", "跨模态推理", "Cross-modal Reasoning", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "InternData-N1: A high-quality navigation dataset with the most diverse scenes and extensive randomization across embodiments/viewpoints, including 3k+ scenes and 830k VLN data. InternData-N1：一个高质量导航数据集，具有最丰富的场景和跨具身/视角的广泛随机化，包含 3000+ 场景和 83 万条 VLN 数据。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": true, "key": "opencompass:2152", "languages": [], "modality": "multimodal", "name": "InternData-N1", "openness": "restricted", "publisher": "InternRobotics", "released": "2025-07-26", "released_reference": {"basis": "dataset_published", "note": "The official dataset license states the release date, before the later directory and model updates.", "source_key": "opencompass:2152", "source_url": "https://huggingface.co/datasets/InternRobotics/InternData-N1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2152-interndata-n1", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InternData-N1", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "视频生成", "物理智能", "Embodied AI", "图像理解", "Image Understanding", "视觉生成", "Visual Generation", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IntPhys2 tests intuitive physics—permanence, immutability, continuity and solidity—using synthetic videos. SOTA models perform around chance (~50 %), far below human level. IntPhys 2，一个用于评估深度学习模型直观物理理解能力的视频基准。它围绕永恒性、不可变性、时空连续性和实体性四个核心原则，测试模型区分可能与不可能事件的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1976", "languages": [], "modality": null, "name": "IntPhys2", "openness": "restricted", "publisher": "FAIR at Meta", "released": "2025-06-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1976-intphys2", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IntPhys2", "unit": null}, {"aliases": [], "categories": ["physics", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "International Physics Olympiad 2025 (theory) comprises all 3 theory problems from the official 2025 IPhO competition. Results are based on blinded human evaluation with guidelines based on the official competition scoring, validated by domain experts.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ipho-2025:muse-spark", "reported_at": "2026-04-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ipho-2025", "languages": [], "modality": "text", "name": "IPhO 2025", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.826, "raw_min": 0.793, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ipho-2025:muse-spark", "reported_date": "2026-04-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500"}, "unit": null}, "slug": "llm-stats-ipho-2025", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "VQA", "Strong Reasoning", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IQBench is a novel benchmark designed to evaluate the fluid intelligence of VisionLanguage Models (VLMs) using standardized visual IQ tests. It consists of 500 manually collected and annotated visual IQ questions covering various domains. IQBench是一个新基准测试，旨在通过标准化视觉智商测试评估视觉语言模型（VLMs）的流体智力。该基准包含500个手动收集和注释的视觉智商问题，涵盖模式识别、类比推理、视觉算术、空间理解等多个领域。与以往仅关注最终答案准确性的基准不同，IQBench强调对模型推理能力的评估，采用双重评估框架：准确性评分和推理评分。实验表明，即使是性能最高的模型（如o4mini、gemini2.5flash和claude3.7sonnet），在3D空间和字母重组任务上也表现出明显不足，凸显了当前VLMs在通用推理能力上的局限性。IQBench为开发更透明、更具认知能力的多模态系统奠定了基础。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1840", "languages": [], "modality": "multimodal", "name": "IQBench", "openness": "unknown", "publisher": "Harvard Medical School, USA Uppsala University,etc", "released": "2025-05-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1840-iqbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IQBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "安全", "Safety", "智能体", "Agent", "task planning", "安全对齐", "Safety Alignment", "任务执行", "Task Execution", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "IS-Bench is the first evaluation benchmark dedicated to assessing the safety of embodied agents during their interaction with home environments. It encompasses 161 high-risk domestic scenarios, spanning 10 hazard categories, including food poisoning, fire, electric shock, and more. IS-Bench 是首个专注于具身智能体与家用环境交互过程安全性的评测基准。它包含 161 个高风险家居场景及相关日常任务，覆盖食物中毒、火灾、触电等 10 大类常见风险。通过贯穿整个交互过程的动态评测框架，全方位评估具身智能体的安全素养。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2091", "languages": [], "modality": "multimodal", "name": "IS-Bench", "openness": "open", "publisher": "OpenTrustLab", "released": "2025-07-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2091-is-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IS-Bench", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "智能体", "Agent", "任务执行", "Task Execution", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ITBench is a benchmark designed to evaluate the performance of AI agents in real-world IT automation tasks. ITBench 是一个旨在评估 AI 智能体在真实世界 IT 自动化任务中表现的基准。它涵盖了站点可靠性工程、合规与安全运营以及财务运营等关键维度，并包含 102 个真实场景。该基准提供了一个开源框架和多种基线智能体实现，并集成了CrewAI等工具，以促进AI驱动的IT自动化发展。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:2077", "languages": [], "modality": null, "name": "ITBench", "openness": "unknown", "publisher": "IBM , University of Illinois at Urbana-Champaign", "released": "2025-02-07", "released_reference": {"basis": "release_announcement", "note": "The official project changelog explicitly dates its initial release of the paper, scenarios, setup tooling, and agents.", "source_key": "opencompass:2077", "source_url": "https://github.com/itbench-hub/ITBench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2077-itbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ITBench", "unit": null}, {"aliases": [], "categories": ["agentic", "tool-use", "engineering"], "collected_at": "2026-08-25T10:41:06Z", "description": "Kubernetes incident root-cause analysis", "evidence_summary": {"document_count": 1, "model_count": 33, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:itbench-aa:976cc8ad-7904-4056-83c5-960181f47d5f", "reported_at": "2024-12-06", "source_url": "https://artificialanalysis.ai/evaluations/itbench-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:itbench-aa", "languages": [], "modality": null, "name": "ITBench-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 33, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.2146892655367, "display_multiplier": 100, "model_count": 33, "model_count_basis": "source_model_id", "numeric_count": 33, "raw_max": 0.562146892655367, "raw_min": 0.00564971751412429, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:itbench-aa:d93edfe8-bf35-49ad-b56e-b18116142a1c", "reported_date": "2026-07-09", "source_url": "https://artificialanalysis.ai/evaluations/itbench-aa"}, "unit": null}, "slug": "artificial-analysis-itbench-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/itbench-aa", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "legal agent", "interactive benchmark", "legal benchmark", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "J1-Bench is an interactive and comprehensive legal benchmark where LLM agents engage in diverse legal scenarios, completing tasks through interactions with various participants under procedural rules. J1-Bench 是一个交互式的综合法律基准，法律智能体在此参与各种法律情景，根据程序规则通过与不同参与者的互动完成指定的法律任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2073", "languages": [], "modality": null, "name": "J1-Bench", "openness": "restricted", "publisher": "Fudan Data Intelligence and Social Computing (Fudan DISC) Lab", "released": "2025-07-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2073-j1-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/J1-Bench", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "NeurIPS 2024", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "JailTrickBench can evaluate the impact of various attack settings on LLM performance, including 8 key factors of implementing jailbreak attacks on LLMs from both target-level and attack-level perspectives. JailTrickBench用于评估LLM应对各种越狱攻击的能力，涵盖从目标级和攻击级2个角度实施越狱攻击的8个关键因素。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1337", "languages": [], "modality": null, "name": "JailTrickBench", "openness": "unknown", "publisher": "The Hong Kong University of Science and Technology", "released": "2024-06-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1337-jailtrickbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JailTrickBench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "JL1-CD is a large-scale, sub-meter, all-inclusive open-source dataset for remote sensing image change detection (CD). It contains 5,000 pairs of 512×512 pixel satellite images with a resolution of 0.5 to 0.75 meters, covering various types of surface changes in multiple regions of China. JL1-CD 是一个大规模、亚米级、全包含的开源遥感影像变化检测（CD）数据集。它包含 5000 对 512×512 像素的卫星影像，分辨率为 0.5 至 0.75 米，覆盖中国多个地区的各种地表变化。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1554", "languages": [], "modality": null, "name": "JL1-CD", "openness": "open", "publisher": "Beijing National Research Center for Information Science and Technology,etc.", "released": "2025-02-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1554-jl1-cd", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JL1-CD", "unit": null}, {"aliases": [], "categories": ["productivity", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Job Bench evaluates AI agents on realistic professional tasks that require multi-step planning, research, and production of work artifacts.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:job-bench:muse-spark-1.1", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:job-bench", "languages": [], "modality": "text", "name": "Job Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 54.7, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.547, "raw_min": 0.334, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:job-bench:muse-spark-1.1", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-job-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "综合能力", "Comprehensive Capability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "JOB-Complex is a challenging benchmark for traditional and learned query optimizers, containing 30 SQL queries and a plan-selection benchmark with nearly 6000 execution plans. JOB-Complex是一个面向传统与学习式查询优化器的挑战性数据库基准，包含30个SQL查询和近6000个执行计划的计划选择基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2130", "languages": [], "modality": null, "name": "JOB-Complex", "openness": "unknown", "publisher": "Technical University of Darmstadt , DFKI", "released": "2025-07-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2130-job-complex", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JOB-Complex", "unit": null}, {"aliases": [], "categories": ["professional"], "collected_at": null, "description": "Professional job tasks aligned with human work; rubric grading moves the number beyond model capability alone.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000jobbench\u0000tencent_hy4_preview\u0000jobbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000jobbench\u0000tencent_hy4_preview\u0000jobbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:jobbench", "languages": [], "modality": null, "name": "JobBench", "openness": "unknown", "publisher": null, "released": "2026-05-25", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:jobbench", "source_url": "https://arxiv.org/abs/2605.26329"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.7, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 61.7, "raw_min": 61.7, "source_reference": {"obs_id": "curated\u0000jobbench\u0000tencent_hy4_preview\u0000jobbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000jobbench\u0000tencent_hy4_preview\u0000jobbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "jobbench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2605.26329", "unit": "percent"}, {"aliases": ["JointAVBench", "Joint AV-Bench"], "categories": ["multimodal"], "collected_at": null, "description": "An ICLR 2026 poster on joint audio-visual reasoning in Omni-LLMs, built with strict audio-video correlation across five cognitive dimensions and four audio information types. Because the question set requires both modalities, vendors that only report vision-only or audio-only figures are not comparable on this instrument. Automated annotation pipeline; manual verification effort is not quantified.", "evidence_summary": {"document_count": null, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:jointavbench", "languages": [], "modality": null, "name": "JointAVBench", "openness": "unknown", "publisher": null, "released": "2025-12-14", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:jointavbench", "source_url": "https://arxiv.org/abs/2512.12772"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "jointavbench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2512.12772", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "知识储备", "Knowledge", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "JudgeBench is a benchmark aimed at evaluating LLM-based judges for objective correctness on challenging response pairs. JudgeBench 是一个旨在评估基于LLM的裁判在具有挑战性的响应对上的客观正确性的基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1583", "languages": [], "modality": null, "name": "JudgeBench", "openness": "open", "publisher": "UC Berkeley, Washington University in St. Louis", "released": "2024-10-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1583-judgebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JudgeBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "图像评估", "视频评估", "多模态模型", "VLM", "视觉生成", "Visual Generation", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "K-Sort Arena employs K-wise comparison, allowing K models to participate in a free-for-all, providing richer information than pairwise comparison. It also designs a matching algorithm based on exploration-exploitation and probabilistic modeling to achieve more efficient and reliable model ranking. 本项目提出K-Sort Arena，采用 K-wise 比较，允许 K 个模型参与自由混战，提供比成对比较更丰富的信息，并设计基于探索-利用的匹配算法和概率建模，从而实现更高效和更可靠的模型排名。目前，K-Sort Arena 已收集几千次高质量投票并构建了全面的模型排行榜，已用于评估几十种最先进的视觉生成模型，包括文生图和文生视频模型。K-Sort Arena已经历数月的项目内测，期间收到来自加州大学伯克利分校, 新加坡国立大学, 卡内基梅隆大学, 斯坦福大学, 普林斯顿大学, 北京大学等数十家机构的专业人员的技术反馈，现已公开线上发布。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "opencompass:2076", "languages": [], "modality": "multimodal", "name": "K-Sort-Arena", "openness": "unknown", "publisher": "UC Berkeley, 中科院自动化所", "released": "2024-09-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2076-k-sort-arena", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/K-Sort-Arena", "unit": null}, {"aliases": [], "categories": ["agents", "code", "systems"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Kernel Bench L3 evaluates agentic GPU kernel optimization across 50 problems. Qwen reports two metrics for this benchmark: median per-problem speedup over the PyTorch eager reference and the fraction of problems faster than torch.compile.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:kernel-bench-l3:qwen3.7-max", "reported_at": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:kernel-bench-l3", "languages": [], "modality": "text", "name": "Kernel Bench L3", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.96, "raw_min": 0.96, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:kernel-bench-l3:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500"}, "unit": null}, "slug": "llm-stats-kernel-bench-l3", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code", "systems"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "KernelBench Hard evaluates agentic GPU kernel optimization on the hardest problem set. Each question is scored by the agent's submitted operator TFLOPs relative to the theoretical peak of the current hardware, with the benchmark score being the average across all questions.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:kernelbench-hard:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:kernelbench-hard", "languages": [], "modality": "text", "name": "KernelBench Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 28.799999999999997, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.288, "raw_min": 0.288, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:kernelbench-hard:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-kernelbench-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code", "systems"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "KernelGen 1P is an OpenAI AI-self-improvement evaluation that measures whether models can write and optimize compute kernels, part of the suite tracking progress toward accelerating internal research.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:kernelgen-1p:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:kernelgen-1p", "languages": [], "modality": "text", "name": "KernelGen 1P", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.1, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.611, "raw_min": 0.224, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:kernelgen-1p:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500"}, "unit": null}, "slug": "llm-stats-kernelgen-1p", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Kimi Claw 24/7 Bench is Moonshot AI's in-house benchmark for evaluating long-horizon agentic performance in persistent, multi-day coworking tasks. It spans 17 professional scenarios across 610 evaluation points, covering software engineering, ML research, recruiting, trading, and marketing tasks executed through the OpenClaw harness.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:kimi-claw-24-7-bench:kimi-k2.7-code", "reported_at": "2026-06-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:kimi-claw-24-7-bench", "languages": [], "modality": "text", "name": "Kimi Claw 24/7 Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 46.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.469, "raw_min": 0.469, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:kimi-claw-24-7-bench:kimi-k2.7-code", "reported_date": "2026-06-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-kimi-claw-24-7-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Kimi Code Bench v2 is Moonshot AI's in-house benchmark for evaluating coding agents on realistic software engineering tasks across 10+ mainstream programming languages and a production tech stack spanning backend services, infrastructure, performance engineering, systems programming, security, frontend development, and ML/data engineering.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:kimi-code-bench-v2:kimi-k2.7-code", "reported_at": "2026-06-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:kimi-code-bench-v2", "languages": [], "modality": "text", "name": "Kimi Code Bench v2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.89999999999999, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.729, "raw_min": 0.62, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:kimi-code-bench-v2:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-kimi-code-bench-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "knowledge", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "KINA is a knowledge-intensive evaluation that measures a model's breadth and depth of factual knowledge across academic and professional domains.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:kina:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:kina", "languages": [], "modality": "text", "name": "KINA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.3, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.483, "raw_min": 0.466, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:kina:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500"}, "unit": null}, "slug": "llm-stats-kina", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "语言", "Language", "多模态模型", "VLM", "OCR与文档理解", "OCR and Document Understanding", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "KITAB-Bench is a Comprehensive Multi-Domain Benchmark for Arabic OCR and Document Understanding,  and spans 36 sub-domains with over 8,809 samples, carefully curated to rigorously evaluate essential skills required for Arabic OCR and document analysis. KITAB-Bench是一个全面多领域阿拉伯文 OCR 和文档理解基准，包含 36 个子领域，超过 8,809 个样本，经过精心挑选，以严格评估阿拉伯 OCR 和文档分析所需的基本技能。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1543", "languages": ["Arabic"], "modality": "multimodal", "name": "KITAB-Bench", "openness": "unknown", "publisher": "Mohamed bin Zayed University of AI;etc.", "released": "2025-02-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1543-kitab-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KITAB-Bench", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "KMMLU-Redux is a reconstructed version of the existing KMMLU, comprising 2,587 problems from Korean National Technical Qualification (KNTQ) exams. KMMLU-Redux是现有 KMMLU 的一个重建版本，包含来自韩国国家技术资格（KNTQ）考试的 2,587 个问题。我们发现了 KMMLU 中的一些关键问题，包括泄露的答案、缺乏清晰度、问题表述不当、符号错误和污染风险。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:2126", "languages": ["Korean"], "modality": null, "name": "KMMLU-Redux", "openness": "restricted", "publisher": "LGA IResearch , Oneline AI", "released": "2025-07-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2126-kmmlu-redux", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KMMLU-Redux", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "KnowLogic is a knowledge-driven synthetic benchmark designed to evaluate the reasoning abilities of large language models (LLMs). It includes 5400 bilingual (Chinese and English) questions across various domains, covering different aspects of commonsense knowledge and logical reasoning. KnowLogic 是一个以知识驱动的合成基准，旨在评估大型语言模型的推理能力（LLMs）。它包含涵盖各个领域、涵盖常识知识和逻辑推理不同方面的 5400 个中英双语问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1635", "languages": ["English", "Chinese"], "modality": null, "name": "KnowLogic", "openness": "restricted", "publisher": "PKU, etc.", "released": "2025-03-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1635-knowlogic", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KnowLogic", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "VQA", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "KOFFVQA is a carefully crafted free-form visual question answering(VQA) benchmark in the Korean language consisting of 275 questions across 10 different tasks. KOFFVQA是一个精心设计的韩语自由形式视觉问答（VQA）基准测试，包含10个不同任务中的275个问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1706", "languages": ["Korean"], "modality": "multimodal", "name": "KOFFVQA", "openness": "unknown", "publisher": "MAUM AI Inc. / Republic of Korea", "released": "2025-03-31", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1706-koffvqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KOFFVQA", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "大语言模型", "LLM", "逻辑推理", "Reasoning", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Knowledge-Orthogonal Reasoning Benchmark (KOR-Bench) encompasses five task categories: Operation, Logic, Cipher, Puzzle, and Counterfactual. KOR-Bench emphasizes the effectiveness of models in applying new rule descriptions to solve novel rule-driven questions. KOR-Bench用于评估大语言模型的推理能力，包括五个任务类别：操作、逻辑、密码、拼图和反事实。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1245", "languages": [], "modality": null, "name": "KOR-Bench", "openness": "unknown", "publisher": "Multimodal Art Projection Research Community", "released": "2024-10-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1245-kor-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KOR-Bench", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "大语言模型", "LLM", "长上下文", "Long Context", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "L-Eval is a comprehensive Long Context Language Models (LCLMs) evaluation suite with 20 sub-tasks, 508 long documents, and over 2,000 human-labeled query-response pairs encompassing diverse question styles, domains, and input length (3k～200k tokens). L-Eval 是一个全面的长上下文语言模型（LCLMs）评估套件，包括 20 个子任务、508 个长文档和超过 2,000 个人工标记的查询-响应对。它涵盖了多种问答风格、领域和输入长度（3,000 至 200,000 个 token）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:541", "languages": [], "modality": null, "name": "L-Eval", "openness": "restricted", "publisher": null, "released": "2023-10-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-541-l-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/L-Eval", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "agents", "biology"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LABBench2 evaluates models on real-world biology research tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:labbench2:gemini-3.7-flash", "reported_at": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:labbench2", "languages": [], "modality": "text", "name": "LABBench2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.821, "raw_min": 0.821, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:labbench2:gemini-3.7-flash", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500"}, "unit": null}, "slug": "llm-stats-labbench2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The LAMBADA evaluates the capabilities of computational models for text understanding by means of a word prediction task. LAMBADA is a collection of narrative passages sharing the characteristic that human subjects are able to guess their last word if they are exposed to the whole passage, but not if they only see the last sentence preceding the target word. To succeed on LAMBADA, computational models cannot simply rely on local context, but must be able to keep track of information in the broader discourse.\nThe LAMBADA dataset is extracted from BookCorpus and consists of 10'022 passages, divided into 4'869 development and 5'153 test passages, comprising 203 million words. LAMBADA 通过一个单词预测任务来评估计算模型对文本理解的能力。LAMBADA 是有如下特点的一组叙述性文章：如果面对整篇文章，人们可以猜测它们的最后一个单词，但如果他们只看到目标单词前面的最后一句话，就无法猜测。为了在 LAMBADA 上由好的效果，模型不能仅仅依赖于局部上下文，而必须能够跟踪更广泛的话语信息。\nLAMBADA 数据集是从 BookCorpus 中提取的，包括 10,022 段落，分为 4,869 个开发段落和 5,153 个测试段落，共计 2.03 亿个单词。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:523", "languages": [], "modality": null, "name": "LAMBADA", "openness": "unknown", "publisher": null, "released": "2016-06-20", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the LAMBADA word-prediction dataset.", "source_key": "opencompass:523", "source_url": "https://arxiv.org/abs/1606.06031"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-523-lambada", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LAMBADA", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "知识储备", "Knowledge", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The benchmark covers questions from three major categories: (1) Arts & Entertainment, (2) Lifestyle & Personal Development, and (3) Society & Culture, encompassing over 45 subcategories in total. 基准旨在评估个性化长篇答案生成。该基准涵盖三大类问题:(1)艺术与娱乐,(2)生活与个人发展,(3)社会与文化,共包含45个以上的子类别。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1924", "languages": [], "modality": null, "name": "LaMP-QA", "openness": "restricted", "publisher": "Center for Intelligent Information Retrieval (CIIR) ,etc.", "released": "2025-05-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1924-lamp-qa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LaMP-QA", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "物理智能", "Embodied AI", "语言理解", "Comprehension", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LangNavBench is a benchmark for evaluating natural language understanding in semantic navigation, built upon the manually verified LangNav open-set dataset. LangNavBench是一个语义导航中自然语言理解评测基准，基于手工验证的LangNav开放集数据集，评测具身智能体在自然语言指令引导下的目标定位能力，涵盖类别层次理解、对象属性识别和空间关系推理等维度.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2129", "languages": [], "modality": null, "name": "LangNavBench", "openness": "unknown", "publisher": "Simon Fraser University , University of Padova , Fondazione Bruno Kessler ,etc", "released": "2025-07-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2129-langnavbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LangNavBench", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "大语言模型", "LLM", "长上下文", "Long Context", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LaRA is a focused benchmark for testing Retrieval-Augmented Generation (RAG) and long-context LLMs. LaRA 是专为评估检索增强生成（RAG）和长上下文大型语言模型（LLM）而打造的基准。它围绕信息定位、片段对比、内容推理和幻觉检测四大能力，通过 2 326 条测试用例，对四类问答任务和三种长文本场景（小说、学术论文、财务报表）进行全面测评。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2089", "languages": [], "modality": null, "name": "LaRA", "openness": "unknown", "publisher": "The Hong Kong University of Science and Technology , Tongyi Lab, etc.", "released": "2025-02-14", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the LaRA benchmark, before its July Hub-card creation.", "source_key": "opencompass:2089", "source_url": "https://arxiv.org/abs/2502.09977"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2089-lara", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LaRA", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LBPP (v2) benchmark - specific documentation not found in official sources, possibly related to language-based planning problems", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:lbpp-(v2):gemini-diffusion", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:lbpp-(v2)", "languages": [], "modality": "text", "name": "LBPP (v2)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.568, "raw_min": 0.568, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:lbpp-(v2):gemini-diffusion", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500"}, "unit": null}, "slug": "llm-stats-lbpp-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LCSTS is a large corpus of Chinese short text summarization dataset constructed from the Chinese microblogging website Sina Weibo, which is released to the public. This corpus consists of over 2 million real Chinese short texts with short summaries given by the author of each text. LCSTS是一个大规模的中文短文本摘要数据集，从中国微博网站新浪微博中构建而成，并已开源。该数据集包含超过 200 万条真实的中文短文本，每个文本都提供了一个简短的摘要。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:520", "languages": ["Chinese"], "modality": null, "name": "LCSTS", "openness": "unknown", "publisher": null, "released": "2015-06-19", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the Chinese short-text summarization corpus.", "source_key": "opencompass:520", "source_url": "https://arxiv.org/abs/1506.05865"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-520-lcsts", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LCSTS", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The Legal Agent Benchmark (LAB) is Harvey's open-source benchmark for evaluating AI agents on complex, long-horizon legal work. Tasks are scored under an all-pass standard against expert-curated rubrics, where a task passes only if every required rubric criterion (facts, conclusions, citations, structure, and analytical moves) passes.", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:legal-agent-benchmark:gemini-3-flash-preview", "reported_at": "2025-12-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:legal-agent-benchmark", "languages": [], "modality": "text", "name": "Legal Agent Benchmark", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 13.3, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.133, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:legal-agent-benchmark:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500"}, "unit": null}, "slug": "llm-stats-legal-agent-benchmark", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500", "unit": null}, {"aliases": ["Legal Agent Benchmark"], "categories": ["legal"], "collected_at": null, "description": "Open-source agent benchmark from Harvey (github.com/harveyai/harvey- labs): 1,200+ tasks across 24 practice areas, all-pass LLM-judge grading. Absolute scores are very low across all models (0-13%), so ordering at the top is decided by a handful of tasks and small gaps are noise.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:legal_agent_benchmark", "languages": [], "modality": null, "name": "Legal Agent Benchmark", "openness": "unknown", "publisher": null, "released": "2026-05-06", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:legal_agent_benchmark", "source_url": "https://www.harvey.ai/blog/introducing-harveys-legal-agent-benchmark"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "legal_agent_benchmark", "source": "model_reports", "source_url": "https://www.harvey.ai/blog/introducing-harveys-legal-agent-benchmark", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "视觉问答", "Visual-Qa", "物理智能", "Embodied AI", "逻辑推理", "图像理解", "Image Understanding", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A multi-level evaluation benchmark of multimodal reasoning with with 3.4K contemporary images and 60K+ human-authored questions covering eight tasks and 12 daily scenarios, forming three progressive task tiers, i.e., perception, understanding, and reasoning. 多层次多模态推理评测基准", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2422", "languages": ["Multilingual"], "modality": "multimodal", "name": "LENS", "openness": "unknown", "publisher": "武汉理工大学，清华大学，中科院自动化所", "released": "2025-06-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2422-lens", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LENS", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "science", "healthcare", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LifeSciBench is an expert-authored, expert-reviewed benchmark of 750 open-ended life-science research tasks spanning seven workflows and seven biological domains. Responses are graded against 19,020 physician- and scientist-written rubric criteria rather than multiple-choice answers, and most tasks require interpreting attached artifacts such as figures, PDFs, and sequence files.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:lifescibench:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:lifescibench", "languages": [], "modality": "multimodal", "name": "LifeSciBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.9, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.599, "raw_min": 0.512, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:lifescibench:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500"}, "unit": null}, "slug": "llm-stats-lifescibench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Aiming at evaluating LLM's advanced reasoning abilities, LingOly is composed of  olympiad-level linguistic reasoning puzzles in low-resource and extinct languages. It covers more than 90 mostly low-resource languages, and contains 1,133 problems across 6 formats and 5 levels of human difficulty. LingOly由低资源和已灭绝语言的奥林匹克级别语言推理谜题组成，用于评估大语言模型的高级推理能力。其中涵盖了90多种语言，共有1133个涉及6种格式和5个人工难度级别的问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1238", "languages": [], "modality": null, "name": "LINGOLY", "openness": "restricted", "publisher": "University of Oxford", "released": "2024-06-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1238-lingoly", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LINGOLY", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "language", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark for multimodal spatial-language understanding and visual-linguistic question answering.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:lingoqa:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:lingoqa", "languages": [], "modality": "multimodal", "name": "LingoQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.39999999999999, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.834, "raw_min": 0.792, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:lingoqa:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500"}, "unit": null}, "slug": "llm-stats-lingoqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.", "evidence_summary": {"document_count": 1, "model_count": 38, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livebench:o1-preview", "reported_at": "2024-09-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:livebench", "languages": [], "modality": "text", "name": "LiveBench", "openness": "restricted", "publisher": "Abacus.AI", "released": "2024-06-12", "released_reference": {"basis": "release_announcement", "note": "The official datasheet states the suite became public on June 12, before the June 27 paper.", "source_key": "opencompass:1246", "source_url": "https://github.com/LiveBench/LiveBench/blob/main/docs/DATASHEET.md"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 38, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.6, "display_multiplier": 100, "model_count": 38, "model_count_basis": "source_model_id", "numeric_count": 38, "raw_max": 0.846, "raw_min": 0.296, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livebench:o3-mini", "reported_date": "2025-01-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500"}, "unit": null}, "slug": "llm-stats-livebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500", "unit": null}, {"aliases": ["LiveBench"], "categories": ["general"], "collected_at": null, "description": "Contents change by release date; each date is a different benchmark.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:livebench", "languages": [], "modality": null, "name": "LiveBench", "openness": "unknown", "publisher": null, "released": "2024-06-06", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:livebench", "source_url": "https://livebench.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "livebench", "source": "model_reports", "source_url": "https://livebench.ai/", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LiveBench contains questions that are based on recently-released math competitions, arXiv papers, news articles, and datasets, and it contains harder, contamination-free versions of tasks from previous benchmarks such as Big-Bench Hard, AMPS, and IFEval. LiveBench是一个LLM基准测试，涵盖数学、编码、推理、语言、指令遵循和数据分析，包含基于最近发布的数学竞赛、arXiv 论文、新闻文章和数据集的问题，及经典基准测试（如 Big-Bench Hard、AMPS 和 IFEval）的更难、无污染的任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1246", "languages": [], "modality": null, "name": "LiveBench", "openness": "restricted", "publisher": "Abacus.AI", "released": "2024-06-12", "released_reference": {"basis": "release_announcement", "note": "The official datasheet states the suite became public on June 12, before the June 27 paper.", "source_key": "opencompass:1246", "source_url": "https://github.com/LiveBench/LiveBench/blob/main/docs/DATASHEET.md"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1246-livebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveBench", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.", "evidence_summary": {"document_count": 1, "model_count": 14, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livebench-20241125:qwen3-235b-a22b-instruct-2507", "reported_at": "2025-07-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livebench-20241125", "languages": [], "modality": "text", "name": "LiveBench 20241125", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 14, "score_direction": "higher_is_better", "score_summary": {"display_max": 79.60000000000001, "display_multiplier": 100, "model_count": 14, "model_count_basis": "source_model_id", "numeric_count": 14, "raw_max": 0.796, "raw_min": 0.609, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livebench-20241125:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500"}, "unit": null}, "slug": "llm-stats-livebench-20241125", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500", "unit": null}, {"aliases": [], "categories": ["coding"], "collected_at": "2026-08-25T10:41:06Z", "description": "Coding", "evidence_summary": {"document_count": 1, "model_count": 343, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:livecodebench:83cb898e-05d9-4e4b-9de3-2d305014d923", "reported_at": "2023-03-14", "source_url": "https://artificialanalysis.ai/evaluations/livecodebench"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:livecodebench", "languages": [], "modality": null, "name": "LiveCodeBench", "openness": "unknown", "publisher": null, "released": "2024-03-12", "released_reference": {"basis": "paper_first_version", "note": "First version of the LiveCodeBench introduction. Retrospective evaluations of older models do not move its release into 2023.", "source_key": "artificial-analysis:livecodebench", "source_url": "https://arxiv.org/abs/2403.07974"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 343, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.7460317460317, "display_multiplier": 100, "model_count": 343, "model_count_basis": "source_model_id", "numeric_count": 343, "raw_max": 0.917460317460317, "raw_min": 0.00211640211640212, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:livecodebench:d1122eff-ee85-4fdc-8a9f-23bee6590667", "reported_date": "2025-11-18", "source_url": "https://artificialanalysis.ai/evaluations/livecodebench"}, "unit": null}, "slug": "artificial-analysis-livecodebench", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/livecodebench", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "evidence_summary": {"document_count": 1, "model_count": 75, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livecodebench:qwen2-7b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:livecodebench", "languages": [], "modality": "text", "name": "LiveCodeBench", "openness": "unknown", "publisher": "University of California, Berkeley", "released": "2024-03-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 75, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.5, "display_multiplier": 100, "model_count": 75, "model_count_basis": "source_model_id", "numeric_count": 75, "raw_max": 0.935, "raw_min": 0.019, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livecodebench:deepseek-v4-pro-max", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500"}, "unit": null}, "slug": "llm-stats-livecodebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500", "unit": null}, {"aliases": ["LiveCodeBench", "LCB"], "categories": ["coding"], "collected_at": null, "description": "Time-sliced problem set. A score without its problem window is not comparable to any other score.", "evidence_summary": {"document_count": 17, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000livecodebench\u0000qwen2_5_coder_report\u0000livecodebench\u0000Pass@1\u0000Qwen2.5-Coder-32B-Instruct", "reported_at": "2024-09-18", "source_url": "https://arxiv.org/abs/2409.12186"}, "first_score_reported_at": "2024-09-18", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000livecodebench\u0000qwen2_5_coder_report\u0000livecodebench\u0000Pass@1\u0000Qwen2.5-Coder-32B-Instruct", "reported_at": "2024-09-18", "source_url": "https://arxiv.org/abs/2409.12186"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:livecodebench", "languages": [], "modality": null, "name": "LiveCodeBench", "openness": "unknown", "publisher": null, "released": "2024-03-12", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:livecodebench", "source_url": "https://livecodebench.github.io/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.5, "display_multiplier": 1, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 93.5, "raw_min": 31.4, "source_reference": {"obs_id": "curated\u0000livecodebench\u0000deepseek_v4_model_card\u0000livecodebench\u0000think max, pass@1\u0000DeepSeek-V4-Pro", "observation_id": "curated\u0000livecodebench\u0000deepseek_v4_model_card\u0000livecodebench\u0000think max, pass@1\u0000DeepSeek-V4-Pro", "reported_at": "2026-04-22", "reported_date": "2026-04-22", "source_id": "deepseek_v4_model_card", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"}, "unit": "percent"}, "slug": "livecodebench", "source": "model_reports", "source_url": "https://livecodebench.github.io/", "unit": "percent"}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LiveCodeBench evaluates LLMs' coding abilities. It continuously collects new problems over time from contests across LeetCode, AtCoder, and CodeForces. Notably, it also focuses on a broader range of code related capabilities besides code generation. LiveCodeBench用于评估大语言模型的代码能力，包含来自LeetCode、AtCoder和CodeForces的动态更新的问题，并在代码生成能力的基础上将更广泛的相关能力纳入考量。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1413", "languages": [], "modality": null, "name": "LiveCodeBench", "openness": "unknown", "publisher": "University of California, Berkeley", "released": "2024-03-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1413-livecodebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveCodeBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveCodeBench Pro is an advanced evaluation benchmark for large language models for code that uses Elo ratings to rank models based on their performance on coding tasks. It evaluates models on real-world coding problems from programming contests (LeetCode, AtCoder, CodeForces) and provides a relative ranking system where higher Elo scores indicate superior performance.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livecodebench-pro:gemini-3-pro-preview", "reported_at": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livecodebench-pro", "languages": [], "modality": "text", "name": "LiveCodeBench Pro", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 2887.0, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 2887.0, "raw_min": 0.8, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livecodebench-pro:gemini-3.1-pro-preview", "reported_date": "2026-02-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-livecodebench-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500", "unit": null}, {"aliases": ["LiveCodeBench Pro", "LCB Pro"], "categories": ["coding"], "collected_at": null, "description": "Reported as an Elo rating against competitive-programming problems, not a pass rate, so it cannot be read on the same axis as LiveCodeBench.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000livecodebench_pro\u0000google_gemini_3_1_pro_model_card\u0000livecodebench_pro\u0000Thinking (High), Elo\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "first_score_reported_at": "2026-02-19", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000livecodebench_pro\u0000google_gemini_3_1_pro_model_card\u0000livecodebench_pro\u0000Thinking (High), Elo\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:livecodebench_pro", "languages": [], "modality": null, "name": "LiveCodeBench Pro", "openness": "unknown", "publisher": null, "released": "2025-06-13", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:livecodebench_pro", "source_url": "https://livecodebenchpro.com/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 2887.0, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 2887.0, "raw_min": 2887.0, "source_reference": {"obs_id": "curated\u0000livecodebench_pro\u0000google_gemini_3_1_pro_model_card\u0000livecodebench_pro\u0000Thinking (High), Elo\u0000Gemini 3.1 Pro", "observation_id": "curated\u0000livecodebench_pro\u0000google_gemini_3_1_pro_model_card\u0000livecodebench_pro\u0000Thinking (High), Elo\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "reported_date": "2026-02-19", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "unit": "elo"}, "slug": "livecodebench_pro", "source": "model_reports", "source_url": "https://livecodebenchpro.com/", "unit": "elo"}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livecodebench-v5:gemini-2.0-flash-lite", "reported_at": "2025-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livecodebench-v5", "languages": [], "modality": "text", "name": "LiveCodeBench v5", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.6, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.756, "raw_min": 0.186, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livecodebench-v5:gemini-2.5-pro", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500"}, "unit": null}, "slug": "llm-stats-livecodebench-v5", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livecodebench-v5-24.12-25.2:kimi-k1.5", "reported_at": "2025-01-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livecodebench-v5-24.12-25.2", "languages": [], "modality": "text", "name": "LiveCodeBench v5 24.12-25.2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.625, "raw_min": 0.625, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livecodebench-v5-24.12-25.2:kimi-k1.5", "reported_date": "2025-01-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500"}, "unit": null}, "slug": "llm-stats-livecodebench-v5-24-12-25-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "evidence_summary": {"document_count": 1, "model_count": 56, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livecodebench-v6:kimi-k2-base", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:livecodebench-v6", "languages": [], "modality": "text", "name": "LiveCodeBench v6", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 56, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.60000000000001, "display_multiplier": 100, "model_count": 56, "model_count_basis": "source_model_id", "numeric_count": 56, "raw_max": 0.916, "raw_min": 0.263, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livecodebench-v6:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500"}, "unit": null}, "slug": "llm-stats-livecodebench-v6", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livecodebench(01-09):deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livecodebench(01-09)", "languages": [], "modality": "text", "name": "LiveCodeBench(01-09)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 41.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.418, "raw_min": 0.418, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livecodebench(01-09):deepseek-v2.5", "reported_date": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500"}, "unit": null}, "slug": "llm-stats-livecodebench-01-09", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "Spoken text", "Long context", "Live streams", "大语言模型", "LLM", "长上下文", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LiveLongBench is the first spoken long-text benchmark designed to address the challenges of long-context understanding in real-world dialogues, characterized by speech-specific features, high redundancy, and uneven information density. Existing benchmarks fail to capture these complexities, limiting LiveLongBench 是首个面向口语长文本理解的基准测试，基于直播内容构建，涵盖检索类、推理类及混合类三种任务类型，针对现实对话中存在的语音特性、高冗余性和信息密度不均等挑战。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1787", "languages": [], "modality": null, "name": "LiveLongBench", "openness": "unknown", "publisher": null, "released": "2025-04-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1787-livelongbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveLongBench", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "数学", "Math", "大语言模型", "LLM", "逻辑推理", "Reasoning", "数理能力", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LiveMathBench can capture LLM capabilities in complex reasoning tasks, including challenging latest question sets from various mathematical competitions. LiveMathBench用于评估大语言模型在复杂推理方面的表现，由极具挑战性的现代数学问题组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1397", "languages": [], "modality": null, "name": "LiveMathBench", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-12-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1397-livemathbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveMathBench", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveMathematicianBench evaluates research-level mathematical reasoning on continuously refreshed, contamination-resistant problems.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livemathematicianbench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livemathematicianbench", "languages": [], "modality": "text", "name": "LiveMathematicianBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 27.700000000000003, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.277, "raw_min": 0.209, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livemathematicianbench:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500"}, "unit": null}, "slug": "llm-stats-livemathematicianbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveSports-3K evaluates fine-grained understanding and commentary of live sports video.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livesports-3k:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livesports-3k", "languages": [], "modality": "multimodal", "name": "LiveSports-3K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.10000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.771, "raw_min": 0.768, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livesports-3k:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500"}, "unit": null}, "slug": "llm-stats-livesports-3k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LiveSQLBench evaluates models on generating correct SQL queries against live PostgreSQL databases. The LiveSQLBench-Base-Full v1 dataset contains 600 questions across 22 PostgreSQL databases, testing real-world database reasoning and query generation.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:livesqlbench:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:livesqlbench", "languages": [], "modality": "text", "name": "LiveSQLBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 40.17, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.4017, "raw_min": 0.4017, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:livesqlbench:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500"}, "unit": null}, "slug": "llm-stats-livesqlbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LLaVA-Bench evaluates the model's ability in more challenging tasks, including a diverse set of 24 images with 60 questions in total, including indoor and outdoor scenes, memes, paintings, sketches, etc. LLaVA-Bench用于评估多模态大模型应对复杂任务的能力，内含24 张图像及60个问题，包括室内和室外场景、模因、绘画、素描等。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1360", "languages": [], "modality": "multimodal", "name": "LLaVA-Bench", "openness": "open", "publisher": "University of Wisconsin–Madison", "released": "2023-04-17", "released_reference": {"basis": "paper_first_version", "note": "The original Visual Instruction Tuning paper introduces the visual-chat evaluation benchmark.", "source_key": "opencompass:1360", "source_url": "https://arxiv.org/abs/2304.08485"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1360-llava-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLaVA-Bench", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "智能体", "Agent", "任务执行", "Task Execution", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LLM-BabyBench is a benchmark suite designed to evaluate Large Language Models (LLMs) on grounded planning and reasoning tasks. LLM-BabyBench是一个专门评估大语言模型在交互环境中规划和推理能力的新基准测试套件。基于BabyAI网格世界的文本适配版本，该基准评估LLMs在三个核心方面的表现：预测动作对环境状态的影响（Predict任务）、生成低级动作序列以实现指定目标（Plan任务）、以及将高级指令分解为连贯的子目标序列（Decompose任务）。\n基准包含16个难度级别，提供三种文本格式（Narrative、Structured、JSON），并配备OmniBot专家代理用于生成基准数据。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1838", "languages": [], "modality": null, "name": "LLM-BabyBench", "openness": "unknown", "publisher": "MBZUAI, Abu Dhabi, UAE", "released": "2025-05-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1838-llm-babybench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLM-BabyBench", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "Strong Reasoning", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "数理能力", "Math", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LLM-SRBench, a comprehensive benchmark with 239 challenging problems across four scientific domains specifically designed to evaluate LLM-based scientific equation discovery methods while preventing trivial memorization. LLM-SRBench，这是一个包含239个挑战性问题的综合基准测试，涵盖四个科学领域，专门设计用于评估基于LLMs的科学方程式发现方法，同时防止简单记忆。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1752", "languages": [], "modality": null, "name": "LLM-SRBench", "openness": "unknown", "publisher": "VinUniversity, etc.", "released": "2025-04-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1752-llm-srbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLM-SRBench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "NeurIPS 2024", "大语言模型", "LLM", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LLM-Uncertainty-Bench is a new benchmarking approach for LLMs that integrates uncertainty quantification. It spans 5 representative natural language processing tasks, each has a dataset with 10,000 instances. LLM-Uncertainty-Bench将不确定性纳入LLM评估，包含5个具有代表性的自然语言处理任务，每个任务都有包含10000个实例的数据集支撑。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1325", "languages": [], "modality": null, "name": "LLM-Uncertainty-Bench", "openness": "restricted", "publisher": "Tencent AI Lab", "released": "2024-01-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1325-llm-uncertainty-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLM-Uncertainty-Bench", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "数学", "Math", "大语言模型", "LLM", "逻辑推理", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LLMThinkBench is a benchmark framework designed to evaluate large language models (LLMs) on basic math reasoning and “overthinking” behaviors, targeting code-executing language models. LLMThinkBench 是一个用于评估大语言模型（LLM）在基础数学推理和“过度思考”行为方面的基准框架，支持对具备代码执行能力的语言模型进行全面评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2047", "languages": [], "modality": null, "name": "LLMThinkBench", "openness": "unknown", "publisher": "DepartmentofComputerScience,VirginiaTech,Blacksburg,VA,USA", "released": "2025-07-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2047-llmthinkbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLMThinkBench", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "AI Language Proficiency Monitor is a multilingual benchmark platform that systematically assesses LLM performance across up to 200 languages. AI Language Proficiency Monitor是一个多语言大语言模型评测平台，系统性评估模型在多达200种语言上的性能表现，特别关注低资源语言，整合FLORES+、MMLU、GSM8K、TruthfulQA和ARC等数据集评测翻译、问答、数学推理和事实性等能力，提供开源自动更新排行榜和交互式仪表板，覆盖全球80-95%人口使用的语言。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2128", "languages": ["Multilingual"], "modality": null, "name": "lm-evaluation-harness", "openness": "unknown", "publisher": "Bundesministerium für wirtschaftliche Zusammenarbeit und Entwicklung,etc", "released": "2025-07-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2128-lm-evaluation-harness", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/lm-evaluation-harness", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "智能体", "Agent", "任务执行", "Task Execution", "长上下文", "Long Context", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LMAct is a benchmark for evaluating frontier multimodal models' in-context imitation learning capabilities in long contexts. LMAct是评估前沿多模态模型在长上下文中的上下文模仿学习能力的基准。它评估了诸如井字棋、国际象棋和雅达利等交互式任务的多模式决策。该测试集包含多达100万个令牌上下文和512个专家演示集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2088", "languages": [], "modality": null, "name": "LMAct", "openness": "unknown", "publisher": "GoogleDeepMind", "released": "2024-12-02", "released_reference": {"basis": "paper_first_version", "note": "First version introducing LMAct; the 2025 revision is not its release.", "source_key": "opencompass:2088", "source_url": "https://arxiv.org/abs/2412.01441"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2088-lmact", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LMAct", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LMArena Text Leaderboard is a blind human preference evaluation benchmark that ranks models based on pairwise comparisons in real-world conversations. The leaderboard uses Elo ratings computed from user preferences in head-to-head model battles, providing a comprehensive measure of overall model capability and style.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:lmarena-text:grok-4.1-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:lmarena-text", "languages": [], "modality": "text", "name": "LMArena Text Leaderboard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 1483.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 1483.0, "raw_min": 1465.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:lmarena-text:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500"}, "unit": null}, "slug": "llm-stats-lmarena-text", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LOCA-Bench is a long-context agentic benchmark. The 256k variant evaluates agents using the official ReAct mode with an environment description length of 256k tokens, measuring how well models reason and act over very long contexts.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:loca-bench-256k:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:loca-bench-256k", "languages": [], "modality": "text", "name": "LOCA-Bench (256k)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 49.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.493, "raw_min": 0.493, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:loca-bench-256k:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500"}, "unit": null}, "slug": "llm-stats-loca-bench-256k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LOKI, a multimodal synthetic data detection benchmark, designed specifically to comprehensively assess the capabilities of LMMs in detecting synthetic data. LOKI是一个多模态合成数据检测基准，专门设计用于全面评估 LMMs 在检测合成数据方面的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1572", "languages": [], "modality": "multimodal", "name": "LOKI", "openness": "open", "publisher": "Shanghai AI Laboratory", "released": "2024-10-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1572-loki", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LOKI", "unit": null}, {"aliases": ["LongBench", "LongBench v2"], "categories": ["long_context"], "collected_at": null, "description": "Nominal context length is not effective context length.", "evidence_summary": {"document_count": 3, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000longbench\u0000qwen3_5_model_card\u0000longbench_v2\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000longbench\u0000qwen3_5_model_card\u0000longbench_v2\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:longbench", "languages": [], "modality": null, "name": "LongBench", "openness": "unknown", "publisher": null, "released": "2023-08-28", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:longbench", "source_url": "https://github.com/THUDM/LongBench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.2, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 63.2, "raw_min": 63.2, "source_reference": {"obs_id": "curated\u0000longbench\u0000qwen3_5_model_card\u0000longbench_v2\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000longbench\u0000qwen3_5_model_card\u0000longbench_v2\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "longbench", "source": "model_reports", "source_url": "https://github.com/THUDM/LongBench", "unit": "percent"}, {"aliases": [], "categories": ["长文本", "Long-Context", "ACL 2024", "大语言模型", "LLM", "长上下文", "Long Context", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LongBench is a benchmark for bilingual, multitask, and comprehensive assessment of long context understanding capabilities of large language models. LongBench 是一个多任务、中英双语、针对大语言模型长文本理解能力的评测基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:542", "languages": ["English", "Chinese", "Multilingual"], "modality": null, "name": "LongBench", "openness": "restricted", "publisher": null, "released": "2023-08-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-542-longbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LongBench", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "structured_output", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LongBench v2 is a benchmark designed to assess the ability of LLMs to handle long-context problems requiring deep understanding and reasoning across real-world multitasks. It consists of 503 challenging multiple-choice questions with contexts ranging from 8k to 2M words across six major task categories: single-document QA, multi-document QA, long in-context learning, long-dialogue history understanding, code repository understanding, and long structured data understanding.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:longbench-v2:deepseek-v3", "reported_at": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:longbench-v2", "languages": [], "modality": "text", "name": "LongBench v2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.3, "display_multiplier": 100, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 0.663, "raw_min": 0.261, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:longbench-v2:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-longbench-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LongCodeBench evaluates the code understanding and comprehension abilities of large language models at very long context windows, scaling up to 1M tokens. It tests whether models can reason about extensive codebases provided in a single prompt by answering multiple-choice questions about the code.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:longcodebench:nova-2-lite", "reported_at": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:longcodebench", "languages": [], "modality": "text", "name": "LongCodeBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.84, "raw_min": 0.84, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:longcodebench:nova-2-lite", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500"}, "unit": null}, "slug": "llm-stats-longcodebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500", "unit": null}, {"aliases": [], "categories": ["factuality", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LongFact evaluates factual precision over long-form generations containing many individual claims. Each claim is extracted and verified, and the model is scored on claim-level precision, measuring whether extended responses introduce unsupported or false statements.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:longfact:mai-thinking-1", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:longfact", "languages": [], "modality": "text", "name": "LongFact", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 98.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.98, "raw_min": 0.98, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:longfact:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500"}, "unit": null}, "slug": "llm-stats-longfact", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:longfact-concepts:gpt-5-2025-08-07", "reported_at": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:longfact-concepts", "languages": [], "modality": "text", "name": "LongFact Concepts", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.7000000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.007, "raw_min": 0.007, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:longfact-concepts:gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500"}, "unit": null}, "slug": "llm-stats-longfact-concepts", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:longfact-objects:gpt-5-2025-08-07", "reported_at": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:longfact-objects", "languages": [], "modality": "text", "name": "LongFact Objects", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.008, "raw_min": 0.008, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:longfact-objects:gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500"}, "unit": null}, "slug": "llm-stats-longfact-objects", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500", "unit": null}, {"aliases": [], "categories": ["image-generation", "language", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LongText-Bench evaluates text-to-image models on their ability to accurately render long text passages within generated images. It includes English (EN) and Chinese (ZH) subsets to assess multilingual text rendering capabilities.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:longtext-bench", "languages": [], "modality": "image", "name": "LongText-Bench", "openness": "unknown", "publisher": null, "released": "2025-07-29", "released_reference": {"basis": "paper_first_version", "note": "The official X-Omni repository and LongText-Bench dataset card identify this paper as their introduction.", "source_key": "llm-stats:longtext-bench", "source_url": "https://arxiv.org/abs/2507.22058"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-longtext-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longtext-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "长文本", "Long-Context", "long video understanding", "omni-modal", "fine-grained temporal understanding", "多模态模型", "VLM", "视频理解", "Video Understanding", "长上下文", "Long Context", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LongVALE is the first long video benchmark integrating fine-grained omni modalities (video, audio, speech) information in videos. It comprises 105K omni-modal events with precise temporal boundaries and detailed omni-modal captions, aiming to advance comprehensive multi-modal video understanding. LongVALE 是首个集成视频中细粒度全模态（视频、音频、语音）信息的长视频理解基准。它包含 10.5 万个具有精确时间边界和详细的全模态描述的事件标注，致力于全面提升多模态视频大语言模型的跨模态推理与细粒度时间感知能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2067", "languages": ["English"], "modality": "multimodal", "name": "LongVALE", "openness": "unknown", "publisher": "Southern University of Science and Technology & University of Birmingham", "released": "2025-04-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2067-longvale", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LongVALE", "unit": null}, {"aliases": [], "categories": ["long_context", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LongVideoBench is a question-answering benchmark featuring video-language interleaved inputs up to an hour long. It includes 3,763 varying-length web-collected videos with subtitles across diverse themes and 6,678 human-annotated multiple-choice questions in 17 fine-grained categories for comprehensive evaluation of long-term multimodal understanding.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:longvideobench:qwen2.5-vl-7b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:longvideobench", "languages": [], "modality": "multimodal", "name": "LongVideoBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.60000000000001, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.806, "raw_min": 0.547, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:longvideobench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500"}, "unit": null}, "slug": "llm-stats-longvideobench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LongVideoBench tests LMMs' understanding of long videos. It's a question-answering benchmark with video-language interleaved inputs up to an hour long and comprises 3,763 web-collected videos with subtitles across diverse themes, LongVideoBench用于评估多模态大模型的长视频理解能力，是基于交错长视频语料构建的问答集，包含3763个不同主题的带字幕的视频。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1510", "languages": [], "modality": "multimodal", "name": "LongVideoBench", "openness": "restricted", "publisher": null, "released": "2024-07-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1510-longvideobench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LongVideoBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "长文本", "Long-Context", "智能体", "Agent", "navigation", "consistency", "物理智能", "Embodied AI", "长上下文", "Long Context", "具身交互", "Embodied Interaction", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A video-action dataset containing many loop-based navigation dataset in Minecraft environment, aiming to boost the spatial consistency and providing insight for the design of memory module 一个视频-动作导航数据集，包括了在Minecraft环境下大量基于回环的导航数据，能够促进世界模型等视频模型空间一致性的训练，启发记忆模块的设计。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1944", "languages": [], "modality": "multimodal", "name": "LoopNav", "openness": "unknown", "publisher": null, "released": "2025-05-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1944-loopnav", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LoopNav", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LSAT (Law School Admission Test) benchmark evaluating complex reasoning capabilities across three challenging tasks: analytical reasoning, logical reasoning, and reading comprehension. The LSAT measures skills considered essential for success in law school including critical thinking, reading comprehension of complex texts, and analysis of arguments.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:lsat:gpt-4-0613", "reported_at": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:lsat", "languages": [], "modality": "text", "name": "LSAT", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.88, "raw_min": 0.88, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:lsat:gpt-4-0613", "reported_date": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500"}, "unit": null}, "slug": "llm-stats-lsat", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "NeurIPS 2024", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LTMbenchmark assess the long-term memory, continual learning, and information integration capabilities of the agents via dynamic conversational tasks. LTMbenchmark通过动态对话任务评估智能体的长期记忆、持续学习和信息集成能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1335", "languages": [], "modality": null, "name": "LTMbenchmark", "openness": "unknown", "publisher": "GoodAI", "released": "2024-09-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1335-ltmbenchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LTMbenchmark", "unit": null}, {"aliases": [], "categories": ["长文本", "long-context", "大语言模型", "LLM", "长上下文", "Long Context", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "LV-Eval is a challenging long-context benchmark with five length levels (16k, 32k, 64k, 128k, and 256k) reaching up to 256k words. The average number of words is 102,380, and the Min/Max number of words is 11,896/387,406. It features two main tasks, single-hop QA and multi-hop QA, comprising 11 bilingual datasets. LV-Eval是一个具备5个长度等级（16k、32k、64k、128k和256k）、最大文本测试长度达到256k的长文本评测基准。LV-Eval的平均文本长度达到102,380字，最小/最大文本长度为11,896/387,406字。LV-Eval主要有两类评测任务——单跳QA和多跳QA，共包含11个涵盖中英文的评测数据子集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:564", "languages": [], "modality": null, "name": "LV-Eval", "openness": "restricted", "publisher": null, "released": "2024-02-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-564-lv-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LV-Eval", "unit": null}, {"aliases": [], "categories": ["long_context", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "LVBench is an extreme long video understanding benchmark designed to evaluate multimodal models on videos up to two hours in duration. It contains 6 major categories and 21 subcategories, with videos averaging five times longer than existing datasets. The benchmark addresses applications requiring comprehension of extremely long videos.", "evidence_summary": {"document_count": 1, "model_count": 25, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:lvbench:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:lvbench", "languages": [], "modality": "multimodal", "name": "LVBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 25, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.39999999999999, "display_multiplier": 100, "model_count": 25, "model_count_basis": "source_model_id", "numeric_count": 25, "raw_max": 0.854, "raw_min": 0.404, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:lvbench:gemini-3.7-flash", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500"}, "unit": null}, "slug": "llm-stats-lvbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "NAACL 2024", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "M3T is a novel benchmark dataset tailored to evaluate NMT systems on the comprehensive task of translating semi-structured documents. This dataset aims to bridge the evaluation gap in document-level NMT systems, acknowledging the challenges posed by rich text layouts in real-world applications. M3T 是旨在评估神经机器翻译（NMT）系统在翻译半结构化文档的综合任务上的表现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1147", "languages": [], "modality": null, "name": "M3T", "openness": "unknown", "publisher": "AWS AI Labs", "released": "2024-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1147-m3t", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/M3T", "unit": null}, {"aliases": [], "categories": ["reasoning", "finance", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Management Consulting Tasks is an internal OpenAI evaluation of long-horizon professional knowledge work drawn from management-consulting workflows, scoring whether models produce correct, decision-ready analyses.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:management-consulting-tasks:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:management-consulting-tasks", "languages": [], "modality": "text", "name": "Management Consulting Tasks (Internal)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 43.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.432, "raw_min": 0.354, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:management-consulting-tasks:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500"}, "unit": null}, "slug": "llm-stats-management-consulting-tasks", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "知识", "Knowledge", "航运", "海运", "大语言模型", "LLM", "知识储备", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MaritimeBench builds a scientific, fair maritime knowledge assessment system. With 1,888 MCQs based on industry standards, we evaluate models' capabilities across shipping domains. MaritimeBench 致力于构建一套科学、公平且严谨的航运知识评估体系。基于行业权威标准，我们持续维护并更新高质量的航运数据集——其中包含1,888道客观选择题（MCQ格式），以全面、多维度地量化模型在航运各领域的能力表现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": true, "key": "opencompass:1694", "languages": [], "modality": null, "name": "MaritimeBench", "openness": "unknown", "publisher": "中远海运科技股份有限公司", "released": "2025-04-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1694-maritimebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MaritimeBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "safety"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MASK is a collection of 1000 questions measuring whether models faithfully report their beliefs when pressured to lie. It operationalizes deception as the rate at which the model lies, i.e., knowingly making false statements intended to be received as true. Lower dishonesty rates indicate better honesty.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mask:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mask", "languages": [], "modality": "text", "name": "MASK", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.51, "raw_min": 0.51, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mask:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500"}, "unit": null}, "slug": "llm-stats-mask", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The MASK evaluation provides a rigorous benchmark for evaluating honesty in large language models by measuring whether models remain truthful when incentivized to lie. The public set contains 1,028 high-quality human-labeled examples across six distinct archetypes. MASK 评估提供了一个严格的基准，用于评估大型语言模型中的诚实度，通过测量模型在受到诱使说谎的激励时是否保持真实性。公共集包含 1,028 个高质量的人标注示例，涵盖六个不同的原型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1609", "languages": [], "modality": null, "name": "MASK", "openness": "restricted", "publisher": "Center for AI Safety, Scale AI", "released": "2025-03-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1609-mask", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MASK", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "VQA", "Embodied Decision Making", "物理智能", "Embodied AI", "逻辑推理", "具身交互", "Embodied Interaction", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Massive-STEPS is a large-scale semantic trajectories dataset designed for understanding and predicting Point-of-Interest (POI) check-ins. Massive-STEPS是一个大规模的语义轨迹数据集，旨在理解和预测兴趣点（POI）签到行为。该数据集基于Semantic Trails数据集构建，覆盖12个全球不同地区的城市，包含2012-2013年和2017-2018年的签到数据，提供了更现代和多样化的POI签到信息。Massive-STEPS不仅丰富了签到数据的语义信息，还通过与Foursquare Open Source Places数据集对齐，增加了POI的地理坐标、名称和地址等元数据。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1847", "languages": [], "modality": "multimodal", "name": "Massive-STEPS", "openness": "open", "publisher": "University of New South Wales", "released": "2025-05-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1847-massive-steps", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Massive-STEPS", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Evaluating Reasoning Capabilities of LLMs Using the Mastermind Board Game. MastermindEval使用猜谜游戏棋盘评估大型语言模型的推理能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1645", "languages": [], "modality": null, "name": "MastermindEval", "openness": "unknown", "publisher": "Humboldt-Universität zu Berlin, DFKI Berlin", "released": "2025-03-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1645-mastermindeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MastermindEval", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects including Prealgebra, Algebra, Number Theory, Counting and Probability, Geometry, Intermediate Algebra, and Precalculus.", "evidence_summary": {"document_count": 1, "model_count": 71, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:math:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:math", "languages": [], "modality": "text", "name": "MATH", "openness": "open", "publisher": null, "released": "2021-11-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 71, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.89999999999999, "display_multiplier": 100, "model_count": 71, "model_count_basis": "source_model_id", "numeric_count": 71, "raw_max": 0.979, "raw_min": 0.124, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:math:o3-mini", "reported_date": "2025-01-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500"}, "unit": null}, "slug": "llm-stats-math", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "大语言模型", "LLM", "数理能力", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MATH is a new dataset of 12,500 challenging competition mathematics problems. Each problem in MATH has a full step-by-step solution. MATH 是一个包含 12,500 个具有挑战性的竞赛数学问题的新数据集。 MATH 中的每个问题都有完整的分步解决方案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:534", "languages": [], "modality": null, "name": "MATH", "openness": "open", "publisher": null, "released": "2021-11-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-534-math", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MATH", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects. This variant uses Chain-of-Thought prompting to encourage step-by-step reasoning.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:math-(cot):llama-3.1-70b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:math-(cot)", "languages": [], "modality": "text", "name": "MATH (CoT)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.0, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.68, "raw_min": 0.519, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:math-(cot):llama-3.1-70b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500"}, "unit": null}, "slug": "llm-stats-math-cot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MATH-500 is a subset of the MATH dataset containing 500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels across seven mathematical subjects including Prealgebra, Algebra, Number Theory, Counting and Probability, Geometry, Intermediate Algebra, and Precalculus.", "evidence_summary": {"document_count": 1, "model_count": 32, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:math-500:o1-mini", "reported_at": "2024-09-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "llm-stats:math-500", "languages": [], "modality": "text", "name": "MATH-500", "openness": "restricted", "publisher": "OpenAI", "released": null, "released_reference": null, "repo_kind": "monorepo_subdir", "repo_resolution_status": "resolved", "score_count": 32, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.2, "display_multiplier": 100, "model_count": 32, "model_count_basis": "source_model_id", "numeric_count": 32, "raw_max": 0.992, "raw_min": 0.6902, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:math-500:longcat-flash-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500"}, "unit": null}, "slug": "llm-stats-math-500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500", "unit": null}, {"aliases": ["MATH-500", "MATH 500", "MATH"], "categories": ["math"], "collected_at": null, "description": "Largely saturated at the frontier.", "evidence_summary": {"document_count": 10, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000math_500\u0000deepseek_v3_report\u0000math_500\u0000EM, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000math_500\u0000deepseek_v3_report\u0000math_500\u0000EM, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:math_500", "languages": [], "modality": null, "name": "MATH-500", "openness": "unknown", "publisher": null, "released": "2021-03-05", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:math_500", "source_url": "https://github.com/openai/prm800k"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 98.0, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 98.0, "raw_min": 90.2, "source_reference": {"obs_id": "curated\u0000math_500\u0000qwen3_technical_report\u0000math_500\u0000thinking mode\u0000Qwen3-235B-A22B (Thinking)", "observation_id": "curated\u0000math_500\u0000qwen3_technical_report\u0000math_500\u0000thinking mode\u0000Qwen3-235B-A22B (Thinking)", "reported_at": "2025-05-14", "reported_date": "2025-05-14", "source_id": "qwen3_technical_report", "source_url": "https://arxiv.org/abs/2505.09388"}, "unit": "percent"}, "slug": "math_500", "source": "model_reports", "source_url": "https://github.com/openai/prm800k", "unit": "percent"}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MathArena Apex is a challenging math contest benchmark featuring the most difficult mathematical problems designed to test advanced reasoning and problem-solving abilities of AI models. It focuses on olympiad-level mathematics and complex multi-step mathematical reasoning.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:matharena-apex:gemini-3-pro-preview", "reported_at": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:matharena-apex", "languages": [], "modality": "text", "name": "MathArena Apex", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.2, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.902, "raw_min": 0.234, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:matharena-apex:deepseek-v4-pro-max", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500"}, "unit": null}, "slug": "llm-stats-matharena-apex", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500", "unit": null}, {"aliases": ["MathArena Apex", "MathArena Apex 2025", "Apex 2025"], "categories": ["math"], "collected_at": null, "description": "The hardest MathArena split; even near-ceiling models sit well below typical math-benchmark numbers.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000matharena_apex_2025\u0000tencent_hy4_preview\u0000matharena_apex_2025\u0000MathArena Apex 2025\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000matharena_apex_2025\u0000tencent_hy4_preview\u0000matharena_apex_2025\u0000MathArena Apex 2025\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:matharena_apex_2025", "languages": [], "modality": null, "name": "MathArena Apex 2025", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.2, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 74.2, "raw_min": 74.2, "source_reference": {"obs_id": "curated\u0000matharena_apex_2025\u0000tencent_hy4_preview\u0000matharena_apex_2025\u0000MathArena Apex 2025\u0000Hy4 preview", "observation_id": "curated\u0000matharena_apex_2025\u0000tencent_hy4_preview\u0000matharena_apex_2025\u0000MathArena Apex 2025\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "matharena_apex_2025", "source": "model_reports", "source_url": "https://matharena.ai/apex", "unit": "percent"}, {"aliases": [], "categories": ["数学", "Math", "大语言模型", "LLM", "数理能力", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MathBench, a new benchmark that rigorously assesses the mathematical capabilities of large\nlanguage models. MathBench spans a wide range of mathematical disciplines, offering a\ndetailed evaluation of both theoretical understanding and practical problem-solving skills. MathBench 严格评估大型语言模型的数学能力。MathBench 涉及广泛的数学学科，提供对理论理解和实际问题解决技能的详细评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1089", "languages": ["English", "Chinese"], "modality": null, "name": "MathBench", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1089-mathbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathBench", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "大语言模型", "LLM", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MathQA is a new large-scale, diverse dataset of 37k English multiple-choice math word problems covering multiple math domain categories by modeling operation programs\ncorresponding to word problems in the AQuA dataset. MathQA 是一个大规模、多样化的数据集，包含 37,000 道英语多项选择数学文字问题，涵盖多个数学领域类别。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1115", "languages": ["English"], "modality": null, "name": "MathQA", "openness": "unknown", "publisher": "Allen Institute for AI", "released": "2019-05-31", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1115-mathqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathQA", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MathVerse evaluates multimodal mathematical reasoning, testing whether models genuinely interpret visual math diagrams rather than relying on text.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mathverse:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mathverse", "languages": [], "modality": "multimodal", "name": "MathVerse", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.7, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.897, "raw_min": 0.892, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mathverse:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500"}, "unit": null}, "slug": "llm-stats-mathverse", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "数学", "Math", "多模态模型", "VLM", "逻辑推理", "数理能力", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MathVerse is intended for evaluating MLLMs' visual math problem-solving, containing 2,612 high-quality, multi-subject math problems with diagrams. MathVerse用于评估多模态大模型的视觉数学问题解决能力，包含2612个高质量、多主题的数学问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1371", "languages": [], "modality": "multimodal", "name": "MathVerse", "openness": "open", "publisher": "The Chinese University of Hong Kong", "released": "2024-03-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1371-mathverse", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathVerse", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MathVerse-Mini is a subset of the MathVerse benchmark for evaluating math reasoning capabilities in vision-language models.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mathverse-mini:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mathverse-mini", "languages": [], "modality": "image", "name": "MathVerse-Mini", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.85, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.85, "raw_min": 0.85, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mathverse-mini:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500"}, "unit": null}, "slug": "llm-stats-mathverse-mini", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MATH-Vision is a dataset designed to measure multimodal mathematical reasoning capabilities. It focuses on evaluating how well models can solve mathematical problems that require both visual understanding and mathematical reasoning, bridging the gap between visual and mathematical domains.", "evidence_summary": {"document_count": 1, "model_count": 33, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mathvision:qvq-72b-preview", "reported_at": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mathvision", "languages": [], "modality": "multimodal", "name": "MathVision", "openness": "open", "publisher": "The Chinese University of Hong Kong", "released": "2024-02-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 33, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.8, "display_multiplier": 100, "model_count": 33, "model_count_basis": "source_model_id", "numeric_count": 33, "raw_max": 0.978, "raw_min": 0.25, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mathvision:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500"}, "unit": null}, "slug": "llm-stats-mathvision", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500", "unit": null}, {"aliases": ["MathVision", "MATH-Vision", "MathVision (mini)"], "categories": ["multimodal"], "collected_at": null, "description": "Prompt-format sensitive: vendors report boxed and unboxed variants and sometimes take the higher of the two.", "evidence_summary": {"document_count": 4, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mathvision\u0000qwen3_5_model_card\u0000mathvision\u0000thinking (temp 0.6, top_p 0.95, top_k 20), fixed boxed prompt\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mathvision\u0000qwen3_5_model_card\u0000mathvision\u0000thinking (temp 0.6, top_p 0.95, top_k 20), fixed boxed prompt\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mathvision", "languages": [], "modality": null, "name": "MathVision", "openness": "unknown", "publisher": null, "released": "2024-02-22", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mathvision", "source_url": "https://mathllm.github.io/mathvision/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.8, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 97.8, "raw_min": 85.6, "source_reference": {"obs_id": "curated\u0000mathvision\u0000moonshot_kimi_k3_model_card\u0000mathvision\u0000reasoning=max, with Python\u0000Kimi K3", "observation_id": "curated\u0000mathvision\u0000moonshot_kimi_k3_model_card\u0000mathvision\u0000reasoning=max, with Python\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "mathvision", "source": "model_reports", "source_url": "https://mathllm.github.io/mathvision/", "unit": "percent"}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "数学", "Math", "多模态模型", "VLM", "逻辑推理", "数理能力", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MathVision measures multimodal mathematical reasoning capabilities through a meticulously curated collection of 3,040 high-quality mathematical problems spanning 16 distinct mathematical disciplines and graded across 5 levels of difficulty. MathVision用于评估多模态大模型的数学推理能力，由涵盖16个数学领域、跨越5个难度级别的3040个高质量数学问题组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1370", "languages": [], "modality": "multimodal", "name": "MathVision", "openness": "open", "publisher": "The Chinese University of Hong Kong", "released": "2024-02-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1370-mathvision", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathVision", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MathVista evaluates mathematical reasoning of foundation models in visual contexts. It consists of 6,141 examples derived from 28 existing multimodal datasets and 3 newly created datasets (IQTest, FunctionQA, and PaperQA), combining challenges from diverse mathematical and visual tasks to assess models' ability to understand complex figures and perform rigorous reasoning.", "evidence_summary": {"document_count": 1, "model_count": 39, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mathvista:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mathvista", "languages": [], "modality": "multimodal", "name": "MathVista", "openness": "open", "publisher": "University of California, Los Angeles", "released": "2023-10-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 39, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.7, "display_multiplier": 100, "model_count": 39, "model_count_basis": "source_model_id", "numeric_count": 39, "raw_max": 0.907, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mathvista:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500"}, "unit": null}, "slug": "llm-stats-mathvista", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500", "unit": null}, {"aliases": ["MathVista"], "categories": ["multimodal"], "collected_at": null, "description": "Mixes OCR quality with visual reasoning.", "evidence_summary": {"document_count": 4, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mathvista\u0000qwen3_5_model_card\u0000mathvista_mini\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mathvista\u0000qwen3_5_model_card\u0000mathvista_mini\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mathvista", "languages": [], "modality": null, "name": "MathVista", "openness": "unknown", "publisher": null, "released": "2023-10-03", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mathvista", "source_url": "https://mathvista.github.io/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.3, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 90.3, "raw_min": 90.3, "source_reference": {"obs_id": "curated\u0000mathvista\u0000qwen3_5_model_card\u0000mathvista_mini\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000mathvista\u0000qwen3_5_model_card\u0000mathvista_mini\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "mathvista", "source": "model_reports", "source_url": "https://mathvista.github.io/", "unit": "percent"}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "多模态", "Multimodal", "数学", "Math", "多模态模型", "VLM", "逻辑推理", "Reasoning", "数理能力", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MathVista is a benchmark designed to combine challenges from diverse mathematical and visual tasks. It consists of 6,141 examples, derived from 28 existing multimodal datasets involving mathematics and 3 newly created datasets. MathVista用于评估多模态大模型的数学能力，结合了丰富的数学和视觉任务挑战。它由6141个示例组成，来自28个涉及数学的现有多模态数据集和3个新创建的数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1178", "languages": [], "modality": "multimodal", "name": "MathVista", "openness": "open", "publisher": "University of California, Los Angeles", "released": "2023-10-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1178-mathvista", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathVista", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MathVista-Mini is a smaller version of the MathVista benchmark that evaluates mathematical reasoning in visual contexts. It consists of examples derived from multimodal datasets involving mathematics, combining challenges from diverse mathematical and visual tasks to assess foundation models' ability to solve problems requiring both visual understanding and mathematical reasoning.", "evidence_summary": {"document_count": 1, "model_count": 24, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mathvista-mini:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:mathvista-mini", "languages": [], "modality": "multimodal", "name": "MathVista-Mini", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 24, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.10000000000001, "display_multiplier": 100, "model_count": 24, "model_count_basis": "source_model_id", "numeric_count": 24, "raw_max": 0.901, "raw_min": 0.5, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mathvista-mini:kimi-k2.5", "reported_date": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500"}, "unit": null}, "slug": "llm-stats-mathvista-mini", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "audio", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MAVERIX (Multimodal Audio-Visual Evaluation Reasoning Index) evaluates multimodal models on tasks that demand tight integration of video and audio information. It features challenges like situational awareness and social sentiment analysis where the answer cannot be reliably determined from a single modality, rigorously testing joint audio-visual understanding.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:maverix:nova-2-omni", "reported_at": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:maverix", "languages": [], "modality": "multimodal", "name": "MAVERIX", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.60000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.666, "raw_min": 0.666, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:maverix:nova-2-omni", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500"}, "unit": null}, "slug": "llm-stats-maverix", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "安全", "Safety", "VQA", "AVQA", "多模态模型", "VLM", "安全对齐", "Safety Alignment", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MAVOS-DD is the first large-scale open-set benchmark for multilingual audio-video deepfake detection. MAVOS-DD是一个大规模的多语言音视频深度伪造检测基准数据集，包含超过250小时的真实和伪造视频，涵盖八种语言。该数据集通过七种不同的深度伪造生成模型生成伪造视频，这些模型基于不同的生成方法，包括说话头像生成、表情转移和换脸。MAVOS-DD设计了多种开放集测试场景，包括开放集模型、开放集语言和全开放集，以评估深度伪造检测模型在未知模型和语言下的泛化能力。实验结果表明，现有的深度伪造检测模型在开放集场景下的性能显著下降，凸显了开发更鲁棒检测技术的必要性。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1848", "languages": ["Multilingual"], "modality": "multimodal", "name": "MAVOS-DD", "openness": "unknown", "publisher": "University of Bucharest,etc", "released": "2025-05-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1848-mavos-dd", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MAVOS-DD", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MAXIFE is a multilingual benchmark evaluating LLMs on instruction following and execution across multiple languages and cultural contexts.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:maxife:qwen3.5-397b-a17b", "reported_at": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:maxife", "languages": [], "modality": "text", "name": "MAXIFE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.2, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.892, "raw_min": 0.392, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:maxife:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500"}, "unit": null}, "slug": "llm-stats-maxife", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality.", "evidence_summary": {"document_count": 1, "model_count": 33, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mbpp:codestral-22b", "reported_at": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:mbpp", "languages": [], "modality": "text", "name": "MBPP", "openness": "unknown", "publisher": null, "released": "2021-08-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 33, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.927, "display_multiplier": 1, "model_count": 33, "model_count_basis": "source_model_id", "numeric_count": 33, "raw_max": 0.927, "raw_min": 0.352, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mbpp:sarvam-30b", "reported_date": "2026-03-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500"}, "unit": null}, "slug": "llm-stats-mbpp", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500", "unit": null}, {"aliases": ["MBPP", "MBPP+"], "categories": ["coding"], "collected_at": null, "description": "Saturated; short single-function tasks.", "evidence_summary": {"document_count": 5, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mbpp\u0000qwen2_5_coder_report\u0000mbpp\u0000MBPP base split\u0000Qwen2.5-Coder-32B-Instruct", "reported_at": "2024-09-18", "source_url": "https://arxiv.org/abs/2409.12186"}, "first_score_reported_at": "2024-09-18", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mbpp\u0000qwen2_5_coder_report\u0000mbpp\u0000MBPP base split\u0000Qwen2.5-Coder-32B-Instruct", "reported_at": "2024-09-18", "source_url": "https://arxiv.org/abs/2409.12186"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:mbpp", "languages": [], "modality": null, "name": "MBPP", "openness": "unknown", "publisher": null, "released": "2021-08-16", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mbpp", "source_url": "https://github.com/google-research/google-research/tree/master/mbpp"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.2, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 90.2, "raw_min": 90.2, "source_reference": {"obs_id": "curated\u0000mbpp\u0000qwen2_5_coder_report\u0000mbpp\u0000MBPP base split\u0000Qwen2.5-Coder-32B-Instruct", "observation_id": "curated\u0000mbpp\u0000qwen2_5_coder_report\u0000mbpp\u0000MBPP base split\u0000Qwen2.5-Coder-32B-Instruct", "reported_at": "2024-09-18", "reported_date": "2024-09-18", "source_id": "qwen2_5_coder_report", "source_url": "https://arxiv.org/abs/2409.12186"}, "unit": "percent"}, "slug": "mbpp", "source": "model_reports", "source_url": "https://github.com/google-research/google-research/tree/master/mbpp", "unit": "percent"}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The benchmark consists of around 1,000 crowd-sourced Python programming problems, designed to be solvable by entry level programmers, covering programming fundamentals, standard library functionality, and so on. Each problem consists of a task description, code solution and 3 automated test cases. 该基准测试由大约1000个入门级程序员可以解决的众包Python编程问题组成，涵盖编程基础知识、标准库功能等。每个问题都由任务描述、代码解决方案和3个自动化测试用例组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:538", "languages": [], "modality": null, "name": "MBPP", "openness": "unknown", "publisher": null, "released": "2021-08-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-538-mbpp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MBPP", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mbpp-++-base-version:llama-3.1-70b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mbpp-++-base-version", "languages": [], "modality": "text", "name": "MBPP ++ base version", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.86, "raw_min": 0.86, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mbpp-++-base-version:llama-3.1-70b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500"}, "unit": null}, "slug": "llm-stats-mbpp-base-version", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mbpp-evalplus:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mbpp-evalplus", "languages": [], "modality": "text", "name": "MBPP EvalPlus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.886, "raw_min": 0.876, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mbpp-evalplus:llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500"}, "unit": null}, "slug": "llm-stats-mbpp-evalplus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mbpp-evalplus-(base):llama-3.1-8b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mbpp-evalplus-(base)", "languages": [], "modality": "text", "name": "MBPP EvalPlus (base)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.728, "raw_min": 0.728, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mbpp-evalplus-(base):llama-3.1-8b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mbpp-evalplus-base", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases. This variant uses pass@1 evaluation metric measuring the percentage of problems solved correctly on the first attempt.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mbpp-pass@1:ministral-8b-instruct-2410", "reported_at": "2024-10-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mbpp-pass@1", "languages": [], "modality": "text", "name": "MBPP pass@1", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.7, "raw_min": 0.7, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mbpp-pass@1:ministral-8b-instruct-2410", "reported_date": "2024-10-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500"}, "unit": null}, "slug": "llm-stats-mbpp-pass-1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases for more rigorous evaluation.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mbpp-plus:mistral-small-3.2-24b-instruct-2506", "reported_at": "2025-06-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mbpp-plus", "languages": [], "modality": "text", "name": "MBPP Plus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.33, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.7833, "raw_min": 0.7833, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mbpp-plus:mistral-small-3.2-24b-instruct-2506", "reported_date": "2025-06-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500"}, "unit": null}, "slug": "llm-stats-mbpp-plus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MBPP+ is an enhanced version of MBPP (Mostly Basic Python Problems) with significantly more test cases (35x) for more rigorous evaluation. MBPP is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mbpp+:qwen-2.5-14b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mbpp+", "languages": [], "modality": "text", "name": "MBPP+", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.1, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.741, "raw_min": 0.402, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mbpp+:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500"}, "unit": null}, "slug": "llm-stats-mbpp-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MCiteBench is a benchmark to evaluate multimodal citation text generation in MLLMs. It consists of 3,000 samples from 1,749 academic papers, featuring 2,000 Explanation tasks and 1,000 Locating tasks, with balanced evidence across text, figures, tables, and mixed modalities. MCiteBench 是一个用于评估多模态大模型中多模态引用文本生成的基准，由 1,749 篇学术论文中的 3,000 个样本组成，包括 2,000 个解释任务和 1,000 个定位任务，在文本、图表、表格和混合模态中平衡证据。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1606", "languages": [], "modality": "multimodal", "name": "MCiteBench", "openness": "restricted", "publisher": "FDU, SHU", "released": "2025-03-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1606-mcitebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MCiteBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MCP Atlas is a benchmark for evaluating AI models on scaled tool use capabilities, measuring how well models can coordinate and utilize multiple tools across complex multi-step tasks.", "evidence_summary": {"document_count": 1, "model_count": 33, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mcp-atlas:claude-opus-4-5-20251101", "reported_at": "2025-11-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:mcp-atlas", "languages": [], "modality": "text", "name": "MCP Atlas", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 33, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.1, "display_multiplier": 100, "model_count": 33, "model_count_basis": "source_model_id", "numeric_count": 33, "raw_max": 0.881, "raw_min": 0.246, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mcp-atlas:muse-spark-1.1", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500"}, "unit": null}, "slug": "llm-stats-mcp-atlas", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500", "unit": null}, {"aliases": ["MCP Atlas", "MCP-Atlas", "MCPAtlas", "MCP-Atlas (Public Set)"], "categories": ["tool_use"], "collected_at": null, "description": "Public and full splits are both reported under one name; the tool server set is part of the measurement.", "evidence_summary": {"document_count": 8, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mcp_atlas\u0000zai_glm_5_model_card\u0000mcp_atlas\u0000think mode, 500-task public subset, 10-min timeout, Gemini 3 Pro judge\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mcp_atlas\u0000zai_glm_5_model_card\u0000mcp_atlas\u0000think mode, 500-task public subset, 10-min timeout, Gemini 3 Pro judge\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mcp_atlas", "languages": [], "modality": null, "name": "MCP Atlas", "openness": "unknown", "publisher": null, "released": "2025-11-18", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mcp_atlas", "source_url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.2, "display_multiplier": 1, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 84.2, "raw_min": 67.8, "source_reference": {"obs_id": "curated\u0000mcp_atlas\u0000moonshot_kimi_k3_model_card\u0000mcp_atlas\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000mcp_atlas\u0000moonshot_kimi_k3_model_card\u0000mcp_atlas\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "mcp_atlas", "source": "model_reports", "source_url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas", "unit": "percent"}, {"aliases": [], "categories": ["agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MCP-Mark evaluates LLMs on their ability to use Model Context Protocol (MCP) tools effectively, testing tool discovery, selection, invocation, and result interpretation across diverse MCP server scenarios.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mcp-mark:deepseek-v3.2", "reported_at": "2025-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mcp-mark", "languages": [], "modality": "text", "name": "MCP-Mark", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.10000000000001, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.811, "raw_min": 0.37, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mcp-mark:kimi-k2.7-code", "reported_date": "2026-06-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500"}, "unit": null}, "slug": "llm-stats-mcp-mark", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MCP-Universe evaluates LLMs on complex multi-step agentic tasks using Model Context Protocol (MCP) tools across diverse interactive environments, testing planning, tool orchestration, and task completion.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mcp-universe:deepseek-v3.2", "reported_at": "2025-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mcp-universe", "languages": [], "modality": "text", "name": "MCP-Universe", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 45.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.459, "raw_min": 0.459, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mcp-universe:deepseek-v3.2", "reported_date": "2025-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500"}, "unit": null}, "slug": "llm-stats-mcp-universe", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500", "unit": null}, {"aliases": ["MCPMark", "MCP-Mark", "MCPMark-Verified"], "categories": ["tool_use"], "collected_at": null, "description": "Depends on live third-party MCP servers, so runs are not reproducible over time.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mcp_mark\u0000qwen3_5_model_card\u0000mcp_mark\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mcp_mark\u0000qwen3_5_model_card\u0000mcp_mark\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mcp_mark", "languages": [], "modality": null, "name": "MCPMark", "openness": "unknown", "publisher": null, "released": "2025-09-30", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mcp_mark", "source_url": "https://mcpmark.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.5, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 94.5, "raw_min": 46.1, "source_reference": {"obs_id": "curated\u0000mcp_mark\u0000moonshot_kimi_k3_model_card\u0000mcpmark_verified\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000mcp_mark\u0000moonshot_kimi_k3_model_card\u0000mcpmark_verified\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "mcp_mark", "source": "model_reports", "source_url": "https://mcpmark.ai/", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MeasureBench evaluates multimodal models on visual measurement and quantitative perception tasks across both real and synthetic imagery, reported as the average over the two settings.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:measurebench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:measurebench", "languages": [], "modality": "multimodal", "name": "MeasureBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.9, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.629, "raw_min": 0.589, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:measurebench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500"}, "unit": null}, "slug": "llm-stats-measurebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "知识", "Knowledge", "Medical", "Agent", "科学智能", "AI for Science", "逻辑推理", "知识储备", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedAgentsBench features 862 challenging medical questions from seven datasets, focusing on cases where models struggle. It emphasizes multi-step clinical reasoning and addresses limitations in existing benchmarks by eliminating simple questions and standardizing evaluation protocols. MedAgentsBench是一个专注于复杂医学推理的基准测试，从七个医学数据集中精选了862个挑战性问题。这些数据集包括MedQA、PubMedQA、MedMCQA、MedBullets、MedExQA、MedXpertQA和MMLU/MMLU-Pro，涵盖了从医学执照考试到研究文献的多种医学问题。\n该基准选择少于50%模型能正确回答的问题，确保医学知识领域全面覆盖，并优先选择需要多步临床推理的问题。这解决了现有评估中简单问题普遍存在、评估协议不一致，以及缺乏性能-成本-时间分析的局限。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1641", "languages": [], "modality": null, "name": "MedAgents-Bench", "openness": "unknown", "publisher": "Yale University", "released": "2025-03-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1641-medagents-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedAgents-Bench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "科学智能", "AI for Science", "语言理解", "Comprehension", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MEDAL is a framework for generating and evaluating multilingual open-domain chatbots and their evaluators. This framework supports various language models and provides a structured approach to creating conversational datasets. MEDAL是一个用于生成和评估多语言开放域聊天机器人及其评估器的框架。该框架支持各种语言模型,并提供了一种结构化的方法来创建对话数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1893", "languages": ["English", "Chinese", "French", "German", "Spanish", "Portuguese", "Multilingual"], "modality": null, "name": "MEDAL", "openness": "restricted", "publisher": "INESC-ID,etc.", "released": "2025-05-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1893-medal", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MEDAL", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "Medical", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedArabiQ introduces a new benchmark dataset consisting of seven Arabic medical tasks, covering multiple specialties and question formats:\n✅ Multiple-choice questions\n✏️ Fill-in-the-blank (with and without choices)\n💬 Patient-doctor question answering 大型语言模型（LLMs）在医疗保健应用中显示出了显著的前景，但由于缺乏高质量的领域特定数据集，它们在阿拉伯语医学领域的表现在很大程度上仍未得到探索。MedArabiQ引入了一个新的基准数据集，包含七个阿拉伯语医学任务，涵盖多个专业和问题格式：\n✅ 多项选择题\n✏️ 填空题（有选项和无选项）\n💬 患者-医生问答\n\n该数据集使用过往医学考试和公开可用资源构建，并进行了修改以评估LLMs在各种能力方面的表现，包括偏见缓解。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1874", "languages": ["Arabic", "Multilingual"], "modality": null, "name": "MedArabiQ", "openness": "unknown", "publisher": "New York University", "released": "2025-05-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1874-medarabiq", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedArabiQ", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "Medical", "科学智能", "AI for Science", "知识储备", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedBench is committed to building a scientific, fair and rigorous Chinese medical model evaluation system and open platform. Based on authoritative standards, we constantly update and maintain high-quality datasets, and comprehensively quantify capabilities of models in various medical dimensions. MedBench致力于打造一个科学、公平且严谨的中文医疗大模型评测体系及开放平台。我们基于医学权威标准，不断更新维护高质量的医学数据集，全方位多维度量化模型在各个医学维度的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1287", "languages": ["Chinese"], "modality": null, "name": "MedBench", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2023-12-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1287-medbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "Medical", "VQA", "MLLM", "科学智能", "AI for Science", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedBookVQA is a medical visual question answering (VQA) benchmark constructed from open-access medical textbooks. It includes 5,000 questions across five clinical task types and is hierarchically organized by imaging modality, anatomical structure, and clinical specialty. MedBookVQA 是一个基于开放获取医学教科书构建的医学视觉问答（VQA）基准数据集。它包含 5,000 个问题，涵盖五种临床任务类型，并按照影像模态、解剖结构和临床专科进行分层组织。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1920", "languages": [], "modality": "multimodal", "name": "MedBookVQA", "openness": "unknown", "publisher": "Hong Kong University of Science and Technology", "released": "2025-05-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1920-medbookvqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedBookVQA", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "Medical", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "the first benchmark that systematically tests an agent’s ability to reliably retrieve and synthesize multi-hop medical facts from live, domain-specific knowledge bases. MedBrowseComp holds 1,000+ human-curated questions that mirror clinical scenarios\n﻿ the first benchmark that systematically tests an agent’s ability to reliably retrieve and synthesize multi-hop medical facts from live, domain-specific knowledge bases. MedBrowseComp holds 1,000+ human-curated questions that mirror clinical scenarios\n﻿", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1832", "languages": [], "modality": null, "name": "MedBrowseComp", "openness": "open", "publisher": "Harvard,etc", "released": "2025-05-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1832-medbrowsecomp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedBrowseComp", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedCalc-Bench focuses on evaluating the medical calculation capability of LLMs. It contains an evaluation set of over 1000 manually reviewed instances from 55 different medical calculation tasks. MedCalc-Bench专注于评估LLM的医学计算能力，包含来自55个不同医学计算任务的1000个经过人工审查的实例。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1241", "languages": [], "modality": null, "name": "MedCalc-Bench", "openness": "unknown", "publisher": "National Institutes of Health", "released": "2024-06-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1241-medcalc-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedCalc-Bench", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MedChemBench is an internal OpenAI evaluation of medicinal-chemistry reasoning, testing whether models can support drug-discovery-relevant chemistry analysis and decision-making.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:medchembench:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:medchembench", "languages": [], "modality": "text", "name": "MedChemBench (Internal)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.3, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.483, "raw_min": 0.304, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:medchembench:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500"}, "unit": null}, "slug": "llm-stats-medchembench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500", "unit": null}, {"aliases": [], "categories": ["医学", "Medical", "幻觉", "Hallucination", "科学智能", "AI for Science", "事实可靠性", "Factual Reliability", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2417", "languages": [], "modality": null, "name": "MedHallTune", "openness": "unknown", "publisher": "The Chinese University of Hong Kong", "released": "2025-02-28", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the medical hallucination instruction-tuning benchmark.", "source_key": "opencompass:2417", "source_url": "https://arxiv.org/abs/2502.20780"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2417-medhalltune", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedHallTune", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "Medical", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedHallu is a comprehensive benchmark dataset designed to evaluate the ability of large language models to detect hallucinations in medical question-answering tasks. MedHallu 旨在评估大型语言模型在医学问题解答任务中检测幻觉的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1548", "languages": ["English"], "modality": null, "name": "MedHallu", "openness": "restricted", "publisher": "University of Texas at Austin, UNC Chapel Hill, Drexel University", "released": "2025-02-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1548-medhallu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedHallu", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "语言生成", "Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedJourney offers a comprehensive assessment of LLMs' effectiveness in real-world clinical settings. It includes multiple tasks from 4 stages of a typical patient's hospital visit journey and comprises 12 datasets. MedJourney用于评估 LLM 在真实临床环境中的有效性，其中包含多个任务，涵盖来自患者就诊典型流程的4个阶段的12个数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1326", "languages": ["Chinese"], "modality": null, "name": "MedJourney", "openness": "unknown", "publisher": "Tencent Youtu Lab", "released": "2024-09-26", "released_reference": {"basis": "paper_first_version", "note": "The clinical-journey benchmark paper's public publication date on OpenReview. This is not the unrelated counterfactual-image MedJourney project.", "source_key": "opencompass:1326", "source_url": "https://openreview.net/forum?id=XXaIoJyYs7"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1326-medjourney", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedJourney", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "NeurIPS 2024", "科学智能", "AI for Science", "安全对齐", "Safety Alignment", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MedSafetyBench is designed to measure the medical safety of LLMs. It includes 1,800 medical safety demonstrations, where each safety demonstration consists of a harmful medical request and a corresponding safe response. MedSafetyBench用于评估LLM在医疗安全上的表现，包含1800个由有害请求和安全响应组成的医疗安全场景。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1331", "languages": [], "modality": null, "name": "MedSafetyBench", "openness": "unknown", "publisher": "Harvard University", "released": "2024-03-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1331-medsafetybench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedSafetyBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive benchmark to evaluate expert-level medical knowledge and advanced reasoning, featuring 4,460 questions spanning 17 specialties and 11 body systems. Includes both text-only and multimodal subsets with expert-level exam questions incorporating diverse medical images and rich clinical information.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:medxpertqa:medgemma-4b-it", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:medxpertqa", "languages": [], "modality": "multimodal", "name": "MedXpertQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.4, "display_multiplier": 100, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 0.784, "raw_min": 0.188, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:medxpertqa:muse-spark", "reported_date": "2026-04-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500"}, "unit": null}, "slug": "llm-stats-medxpertqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "推理", "Reasoning", "知识", "Knowledge", "Medical", "ICML 2025", "科学智能", "AI for Science", "知识储备", "逻辑推理", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "We introduce MedXpertQA, a highly challenging and comprehensive medical benchmark to evaluate expert-level medical knowledge and advanced reasoning. \nIt has been accepted by ICML 2025 and selected by Google DeepMind as the benchmark for MedGemma. MedXpertQA是由清华大学和上海人工智能实验室构建的全面且具有高度挑战性的医学基准，用于评估专家级的医学知识和高级推理能力。论文已被ICML 2025接收，并被Google DeepMind使用作为MedGemma的评估基准。MedXpertQA 共包含 4,460 道题目，涵盖 17 个医学专科和 11 个身体系统。该基准包含两个子集：用于文本医学能力评估的 Text 子集，以及用于多模态医学能力评估的 MM 子集。MM 子集首次引入了带有多样化图像和丰富临床信息（如病历和检查结果）的专家级考试题，区别于传统多模态医学基准中基于图像描述生成的简单问答对。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1891", "languages": [], "modality": null, "name": "MedXpertQA", "openness": "open", "publisher": "清华大学，上海人工智能实验室", "released": "2025-02-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1891-medxpertqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedXpertQA", "unit": null}, {"aliases": [], "categories": ["medical", "multimodal", "knowledge", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MedXpertQA-MM is the multimodal subset of MedXpertQA, evaluating expert-level medical question answering grounded in medical images.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:medxpertqa-mm:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:medxpertqa-mm", "languages": [], "modality": "multimodal", "name": "MedXpertQA-MM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.71, "raw_min": 0.71, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:medxpertqa-mm:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500"}, "unit": null}, "slug": "llm-stats-medxpertqa-mm", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MLQA as part of the MEGA (Multilingual Evaluation of Generative AI) benchmark suite. A multi-way aligned extractive QA evaluation benchmark for cross-lingual question answering across 7 languages (English, Arabic, German, Spanish, Hindi, Vietnamese, and Simplified Chinese) with over 12K QA instances in English and 5K in each other language.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mega-mlqa:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mega-mlqa", "languages": [], "modality": "text", "name": "MEGA MLQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.3, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.653, "raw_min": 0.617, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mega-mlqa:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500"}, "unit": null}, "slug": "llm-stats-mega-mlqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TyDi QA as part of the MEGA benchmark suite. A question answering dataset covering 11 typologically diverse languages (Arabic, Bengali, English, Finnish, Indonesian, Japanese, Korean, Russian, Swahili, Telugu, and Thai) with 204K question-answer pairs. Features realistic information-seeking questions written by people who want to know the answer but don't know it yet.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mega-tydi-qa:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mega-tydi-qa", "languages": [], "modality": "text", "name": "MEGA TyDi QA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.10000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.671, "raw_min": 0.622, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mega-tydi-qa:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500"}, "unit": null}, "slug": "llm-stats-mega-tydi-qa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500", "unit": null}, {"aliases": [], "categories": ["language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Universal Dependencies POS tagging as part of the MEGA benchmark suite. A multilingual part-of-speech tagging dataset based on Universal Dependencies treebanks, utilizing the universal POS tag set of 17 tags across 38 diverse languages from different language families. Used for evaluating multilingual POS tagging systems.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mega-udpos:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mega-udpos", "languages": [], "modality": "text", "name": "MEGA UDPOS", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.604, "raw_min": 0.465, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mega-udpos:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500"}, "unit": null}, "slug": "llm-stats-mega-udpos", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "XCOPA (Cross-lingual Choice of Plausible Alternatives) as part of the MEGA benchmark suite. A typologically diverse multilingual dataset for causal commonsense reasoning in 11 languages, including resource-poor languages like Eastern Apurímac Quechua and Haitian Creole. Requires models to select which choice is the effect or cause of a given premise.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mega-xcopa:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mega-xcopa", "languages": [], "modality": "text", "name": "MEGA XCOPA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.766, "raw_min": 0.631, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mega-xcopa:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500"}, "unit": null}, "slug": "llm-stats-mega-xcopa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "XStoryCloze as part of the MEGA benchmark suite. A cross-lingual story completion task that consists of professionally translated versions of the English StoryCloze dataset to 10 non-English languages. Requires models to predict the correct ending for a given four-sentence story, evaluating commonsense reasoning and narrative understanding.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mega-xstorycloze:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mega-xstorycloze", "languages": [], "modality": "text", "name": "MEGA XStoryCloze", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.828, "raw_min": 0.735, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mega-xstorycloze:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500"}, "unit": null}, "slug": "llm-stats-mega-xstorycloze", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "psychology", "creativity"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MELD (Multimodal EmotionLines Dataset) is a multimodal multi-party dataset for emotion recognition in conversations. Contains approximately 13,000 utterances from 1,433 dialogues extracted from the TV series Friends. Each utterance is annotated with emotion (Anger, Disgust, Sadness, Joy, Neutral, Surprise, Fear) and sentiment labels across audio, visual, and textual modalities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:meld:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:meld", "languages": [], "modality": "multimodal", "name": "Meld", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.99999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.57, "raw_min": 0.57, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:meld:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500"}, "unit": null}, "slug": "llm-stats-meld", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "MER", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MER-UniBench is a benchmark designed to evaluate multimodal large language models (MLLMs) for their emotion understanding capabilities across typical multimodal emotion recognition (MER) tasks. MER-UniBench 是一个旨在评估多模态大语言模型（MLLM）在典型多模态情感识别（MER）任务中情感理解能力的评测基准。它涵盖了细粒度情感识别、基本情感识别和情感分析三个主要维度。该基准利用了包括 MER-Caption 在内的多个数据集，其中 MER-Caption 拥有超过 2000 种细粒度情感类别和 11.5 万个样本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2079", "languages": [], "modality": "multimodal", "name": "MER-UniBench", "openness": "restricted", "publisher": "Institute of Automation, Chinese Academy of Sciences , CMVS, etc.", "released": "2025-01-27", "released_reference": {"basis": "paper_first_version", "note": "The AffectGPT paper introduces MER-UniBench; the older MER-Caption paper is a different artifact.", "source_key": "opencompass:2079", "source_url": "https://arxiv.org/abs/2501.16566"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2079-mer-unibench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MER-UniBench", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Meta's internal evaluation of coding-agent performance.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:meta-internal-coding-bench:muse-spark-1.2", "reported_at": "2026-08-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:meta-internal-coding-bench", "languages": [], "modality": "text", "name": "Meta Internal Coding Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.706, "raw_min": 0.706, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:meta-internal-coding-bench:muse-spark-1.2", "reported_date": "2026-08-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-meta-internal-coding-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MEWC is a benchmark that evaluates AI model performance on multi-environment web challenges, testing agents' ability to navigate and complete complex tasks across diverse web environments.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mewc:minimax-m2.5", "reported_at": "2026-02-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mewc", "languages": [], "modality": "text", "name": "MEWC", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.744, "raw_min": 0.744, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mewc:minimax-m2.5", "reported_date": "2026-02-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500"}, "unit": null}, "slug": "llm-stats-mewc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MGSM (Multilingual Grade School Math) is a benchmark of grade-school math problems. Contains 250 grade-school math problems manually translated from the GSM8K dataset into ten typologically diverse languages: Spanish, French, German, Russian, Chinese, Japanese, Thai, Swahili, Bengali, and Telugu. Evaluates multilingual mathematical reasoning capabilities.", "evidence_summary": {"document_count": 1, "model_count": 31, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mgsm:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mgsm", "languages": [], "modality": "text", "name": "MGSM", "openness": "open", "publisher": null, "released": "2022-10-06", "released_reference": null, "repo_kind": "monorepo_subdir", "repo_resolution_status": "resolved", "score_count": 31, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.30000000000001, "display_multiplier": 100, "model_count": 31, "model_count_basis": "source_model_id", "numeric_count": 31, "raw_max": 0.923, "raw_min": 0.479, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mgsm:llama-4-maverick", "reported_date": "2025-04-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500"}, "unit": null}, "slug": "llm-stats-mgsm", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "instruction_following", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MIABench evaluates multimodal instruction alignment and following capabilities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:miabench:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:miabench", "languages": [], "modality": "multimodal", "name": "MIABench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.927, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.927, "raw_min": 0.927, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:miabench:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500"}, "unit": null}, "slug": "llm-stats-miabench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MIB is a benchmark designed to evaluate mechanistic interpretability methods for neural language models. MIB 是一个旨在评估神经网络语言模型中机械可解释性方法的基准。它主要评估方法在精确地定位因果路径（电路定位）和特定概念（因果变量定位）方面的能力，侧重于忠实度和最小化电路规模。该基准涵盖了IOI和算术等四个任务，并评估了Llama-3.1和Gemma-2等四种模型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2085", "languages": [], "modality": null, "name": "MIB", "openness": "unknown", "publisher": "Boston University , Pr(AI) RGroup , Ai2 , Technion ,etc.", "released": "2025-04-17", "released_reference": {"basis": "paper_first_version", "note": "MIB's own introduction, not the earlier interpretability work cited in its repository.", "source_key": "opencompass:2085", "source_url": "https://arxiv.org/abs/2504.13151"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2085-mib", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIB", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "科学智能", "AI for Science", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MicroVQA is a benchmark that evaluates LLM reasoning on multiple-choice questions about microscopy images, created by expert biologists. MicroVQA，一个评估关于显微镜图像的多选题推理基准，由专家生物学家创建，旨在反映生物研究中能够有意义地协助的任务，每个问题都需要多模态推理。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1667", "languages": [], "modality": "multimodal", "name": "MicroVQA", "openness": "open", "publisher": "Stanford University, Tsinghua University, etc.", "released": "2025-03-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1667-microvqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MicroVQA", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "检索能力", "Retrieval", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Massive Image Embedding Benchmark (MIEB) is to evaluate the performance of image and image-text embedding models across the broadest spectrum to date. MIEB用于评估图像和图像文本嵌入模型在迄今为止最广泛的范围内的性能。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1781", "languages": ["Chinese", "German", "Multilingual"], "modality": "multimodal", "name": "MIEB", "openness": "unknown", "publisher": "Durham University，Zendesk，etc.", "released": "2025-04-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1781-mieb", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIEB", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MiLiC-Eval is an NLP evaluation suite for Minority Languages in China, covering Tibetan (bo), Uyghur (ug), Kazakh (kk, in the Kazakh Arabic script), and Mongolian (mn, in the traditional Mongolian script). MiLiC-Eval 是针对中国少数民族语言的 NLP 评估套件，涵盖藏语（bo）、维吾尔语（ug）、哈萨克语（kk，使用哈萨克阿拉伯文脚本）和蒙古语（mn，使用传统蒙古文脚本）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1658", "languages": ["Arabic", "Multilingual"], "modality": null, "name": "MiLiC-Eval", "openness": "unknown", "publisher": "PKU", "released": "2025-03-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1658-milic-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MiLiC-Eval", "unit": null}, {"aliases": [], "categories": ["multimodal", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MIMIC-CXR is a large publicly available dataset of chest radiographs with free-text radiology reports. Contains 377,110 images corresponding to 227,835 radiographic studies from 65,379 patients at Beth Israel Deaconess Medical Center. The dataset is de-identified and widely used for medical imaging research, automated report generation, and medical AI development.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mimic-cxr:medgemma-4b-it", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mimic-cxr", "languages": [], "modality": "multimodal", "name": "MIMIC CXR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.889, "raw_min": 0.889, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mimic-cxr:medgemma-4b-it", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500"}, "unit": null}, "slug": "llm-stats-mimic-cxr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MiMo Coding Bench evaluates coding-agent capabilities on software engineering tasks reported with the MiMo model family.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mimo-coding-bench:mimo-v2.5", "reported_at": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mimo-coding-bench", "languages": [], "modality": "text", "name": "MiMo Coding Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.7, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.737, "raw_min": 0.718, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mimo-coding-bench:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-mimo-coding-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MIND2WEB is the first dataset for developing and evaluating generalist agents for the web that can follow language instructions to complete complex tasks on any website. MIND2WEB 是首个用于开发和评估通用网页代理的数据集，能够根据语言指令在任何网站上完成复杂任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1125", "languages": [], "modality": null, "name": "Mind2Web", "openness": "unknown", "publisher": "The Ohio State University", "released": "2023-12-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1125-mind2web", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Mind2Web", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Minerva is a benchmark for complex video reasoning, evaluating models on multi-step reasoning over long and information-dense video content.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:minerva:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:minerva", "languages": [], "modality": "multimodal", "name": "Minerva", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.7, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.707, "raw_min": 0.659, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:minerva:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500"}, "unit": null}, "slug": "llm-stats-minerva", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "数学", "Math", "科学智能", "AI for Science", "逻辑推理", "Reasoning", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A context-rich benchmark for evaluating neural theorem proving in realistic scenarios, providing premises, full context, multi-source benchmark, and temporal splits, enabling evaluation of a model's ability to work with context that evolves over time. 一个用于评估现实场景中神经定理证明的丰富上下文基准。通过提供前提、完整上下文、多源基准和时序分割，突出其评估模型处理随时间演变的上下文能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1518", "languages": [], "modality": null, "name": "miniCTX", "openness": "open", "publisher": "Carnegie Mellon University", "released": "2024-08-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1518-minictx", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/miniCTX", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "语言", "Language", "长文本", "Long-Context", "大语言模型", "LLM", "语言理解", "Comprehension", "长上下文", "Long Context", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MiniLongBench is a low-cost benchmark for evaluating the Long Context Understanding (LCU) capabilities of LLMs, featuring a compact yet diverse test set of only 237 samples spanning 6 major task categories and 21 distinct tasks. MiniLongBench is a low-cost benchmark for evaluating the Long Context Understanding (LCU) capabilities of LLMs, featuring a compact yet diverse test set of only 237 samples spanning 6 major task categories and 21 distinct tasks.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1853", "languages": [], "modality": null, "name": "MiniLongBench", "openness": "open", "publisher": "MilkThink-Lab", "released": "2025-05-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1853-minilongbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MiniLongBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "语言", "Language", "VQA", "Strong Reasoning", "多模态模型", "VLM", "检索能力", "Retrieval", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MIRACL-VISION is a large-scale, multilingual visual document retrieval benchmark built by the NVIDIA team, extending the popular MIRACL multilingual text retrieval benchmark. It covers 18 languages and contains 211 original questions. MIRACL-VISION是一个大规模的多语言视觉文档检索基准测试，由NVIDIA团队构建，扩展了流行的MIRACL多语言文本检索基准。该数据集覆盖18种语言，包含211个原创问题，涵盖边界层分析、WKB方法、非线性偏微分方程的渐近解和振荡积分的渐近性等核心主题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1844", "languages": ["English", "Chinese", "Japanese", "Korean", "French", "German", "Spanish", "Arabic", "Russian", "Thai", "Indonesian", "Hindi", "Multilingual"], "modality": "multimodal", "name": "MIRACL-VISION", "openness": "unknown", "publisher": "NVIDIA", "released": "2025-05-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1844-miracl-vision", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIRACL-VISION", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "ACL 2024", "科学智能", "AI for Science", "检索能力", "Retrieval", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MIRAGE is a first-of-its-kind benchmark including 7,663 questions from five medical QA datasets. MIRAGE 是首个此类基准，包含来自五个医学问答数据集的 7,663 个问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": true, "key": "opencompass:1142", "languages": [], "modality": null, "name": "MIRAGE", "openness": "unknown", "publisher": "Univeristy of Virginia", "released": "2024-08-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1142-mirage", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIRAGE", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MJ-Bench incorporates a comprehensive preference dataset to evaluate multimodal judges in providing feedback for image generation models across four key perspectives: alignment, safety, image quality, and bias. MJ-Bench，它包含了一个综合的偏好数据集，用于从四个关键角度评估多模态评委在为图像生成模型提供反馈方面的能力：对齐、安全性、图像质量和偏见。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1623", "languages": [], "modality": "multimodal", "name": "MJ-Bench", "openness": "open", "publisher": "UNC-Chapel Hill, University of Chicago, Stanford University", "released": "2024-07-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1623-mj-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MJ-Bench", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MKQA is an open-domain question answering evaluation set comprising 10k question-answer pairs aligned across 26 typologically diverse languages (260k question-answer pairs in total). MKQA 是一个开放域问答评估集，包含 10,000 对问题和答案，涵盖 26 种类型多样的语言（总计 260,000 对问题和答案）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1099", "languages": [], "modality": null, "name": "MKQA", "openness": "restricted", "publisher": "Apple", "released": "2021-08-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1099-mkqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MKQA", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-25T10:41:06Z", "description": "MLCR-AA overall score", "evidence_summary": {"document_count": 1, "model_count": 71, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:mlcr-aa:b5c1c91a-7474-4409-9a9c-9c2ac45d9eb6", "reported_at": "2024-07-18", "source_url": "https://artificialanalysis.ai/evaluations/mlcr-aa"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:mlcr-aa", "languages": [], "modality": null, "name": "MLCR-AA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 71, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.4444444444444, "display_multiplier": 100, "model_count": 71, "model_count_basis": "source_model_id", "numeric_count": 71, "raw_max": 0.644444444444444, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:mlcr-aa:cd55210d-358e-4df1-ba9c-9acb5f186cc9", "reported_date": "2026-06-09", "source_url": "https://artificialanalysis.ai/evaluations/mlcr-aa"}, "unit": null}, "slug": "artificial-analysis-mlcr-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/mlcr-aa", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MLE-Bench evaluates AI agents on machine learning engineering tasks by measuring their performance on Kaggle competitions.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mle-bench:gemini-3.5-flash-lite", "reported_at": "2026-07-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mle-bench", "languages": [], "modality": "text", "name": "MLE-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.9, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.639, "raw_min": 0.392, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mle-bench:gemini-3.6-flash", "reported_date": "2026-07-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-mle-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500", "unit": null}, {"aliases": ["MLE-Bench", "MLE-bench Revised", "MLE-bench Lite"], "categories": ["ai_research"], "collected_at": null, "description": "Kaggle-derived ML engineering tasks; wall-clock budget and hardware are part of the result. Published by OpenAI.", "evidence_summary": {"document_count": 2, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mle_bench", "languages": [], "modality": null, "name": "MLE-bench", "openness": "unknown", "publisher": null, "released": "2024-10-09", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mle_bench", "source_url": "https://openai.com/index/mle-bench/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "mle_bench", "source": "model_reports", "source_url": "https://openai.com/index/mle-bench/", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MLE-Bench Lite evaluates AI agents on machine learning engineering tasks, testing their ability to build, train, and optimize ML models for Kaggle-style competitions in a lightweight evaluation format.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mle-bench-lite:minimax-m2.7", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mle-bench-lite", "languages": [], "modality": "text", "name": "MLE-Bench Lite", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.60000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.666, "raw_min": 0.666, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mle-bench-lite:minimax-m2.7", "reported_date": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500"}, "unit": null}, "slug": "llm-stats-mle-bench-lite", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MLRC-Bench, a benchmark designed to quantify how effectively language agents can tackle challenging Machine Learning (ML) Research Competitions. MLRC-Bench旨在量化大模型代理如何有效地应对具有挑战性的机器学习 （ML） 研究竞赛。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1779", "languages": [], "modality": null, "name": "MLRC-Bench", "openness": "unknown", "publisher": "University of Michigan, Ann Arbor，LG AI Research，University of Illinois, Chicago", "released": "2025-04-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1779-mlrc-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MLRC-Bench", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MLS-Bench Lite is the official 30-task subset of MLS-Bench for evaluating whether AI systems can invent generalizable and scalable machine learning methods across LLM pretraining and post-training, robotics, world models, computer vision, reinforcement learning, optimization, ML systems, and AI for Science.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mls-bench-lite:kimi-k2.7-code", "reported_at": "2026-06-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mls-bench-lite", "languages": [], "modality": "text", "name": "MLS-Bench Lite", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 48.3, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.483, "raw_min": 0.351, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mls-bench-lite:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500"}, "unit": null}, "slug": "llm-stats-mls-bench-lite", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive benchmark for multi-task long video understanding that evaluates multimodal large language models on videos ranging from 3 minutes to 2 hours across 9 distinct tasks including reasoning, captioning, recognition, and summarization.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mlvu:qwen2.5-vl-7b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mlvu", "languages": [], "modality": "multimodal", "name": "MLVU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.4, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.874, "raw_min": 0.702, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mlvu:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500"}, "unit": null}, "slug": "llm-stats-mlvu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MLVU test MLLMs' understanding of multi-task long videos, encompassing diverse tasks based on various long videos. MLVU用于评估多模态大模型的长视频理解能力，包含面向各种类型长视频的多样化的评估任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1512", "languages": [], "modality": "multimodal", "name": "MLVU", "openness": "restricted", "publisher": "BAAI", "released": "2024-06-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1512-mlvu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MLVU", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MLVU-M benchmark", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mlvu-m:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mlvu-m", "languages": [], "modality": "text", "name": "MLVU-M", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.1, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.821, "raw_min": 0.746, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mlvu-m:qwen3-vl-32b-instruct", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500"}, "unit": null}, "slug": "llm-stats-mlvu-m", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "structured_output"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A challenging multimodal instruction-following benchmark that includes both compose-level constraints for output responses and perception-level constraints tied to input images, with comprehensive evaluation pipeline.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mm-if-eval:pixtral-12b-2409", "reported_at": "2024-09-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mm-if-eval", "languages": [], "modality": "multimodal", "name": "MM IF-Eval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.606, "raw_min": 0.527, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mm-if-eval:lfm-2.5-vl-3b", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500"}, "unit": null}, "slug": "llm-stats-mm-if-eval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A benchmark for evaluating MLLMs' alignment with human preferences. It includes 252 high-quality, human-annotated samples with diverse image types and open-ended questions. Modeled after Arena-style benchmarks, it uses GPT-4o as the judge model and Claude-Sonnet-3 as the reference model. 用于评估 MLLM 与人类偏好的一致性的基准。它包含 252 个高质量、人类标注的样本，具有不同的图像类型和开放式问题。它仿照 Arena 风格的基准，使用 GPT-4o 作为评判模型，Claude-Sonnet-3 作为参考模型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1558", "languages": [], "modality": "multimodal", "name": "MM-AlignBench", "openness": "unknown", "publisher": "Shanghai Jiaotong University,Shanghai AI Laboratory,etc.", "released": "2025-02-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1558-mm-alignbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-AlignBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "search", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MM-BrowserComp evaluates multimodal agents on web browsing and information retrieval tasks, testing a model's ability to perceive, navigate, and extract information from real web environments.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mm-browsercomp:mimo-v2-omni", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mm-browsercomp", "languages": [], "modality": "multimodal", "name": "MM-BrowserComp", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 52.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.52, "raw_min": 0.52, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mm-browsercomp:mimo-v2-omni", "reported_date": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500"}, "unit": null}, "slug": "llm-stats-mm-browsercomp", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MM-ClawBench evaluates models on MiniMax's Claw-style agent benchmark, measuring practical agentic task completion quality in real-world OpenClaw usage scenarios.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mm-clawbench:minimax-m2.7", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mm-clawbench", "languages": [], "modality": "text", "name": "MM-ClawBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.627, "raw_min": 0.627, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mm-clawbench:minimax-m2.7", "reported_date": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500"}, "unit": null}, "slug": "llm-stats-mm-clawbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MM-IQ is a comprehensive evaluation framework comprising 2,710 meticulously curated test items spanning 8 distinct reasoning paradigms. MM-IQ，这是一个包含 2,710 个精心挑选的测试项目的综合评估框架，涵盖了 8 种不同的推理范式。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1566", "languages": [], "modality": "multimodal", "name": "MM-IQ", "openness": "open", "publisher": "Tencent Hunyuan Team", "released": "2025-02-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1566-mm-iq", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-IQ", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "frontend_development", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multimodal web navigation benchmark comprising 2,000 open-ended tasks spanning 137 websites across 31 domains. Each task includes HTML documents paired with webpage screenshots, action sequences, and complex web interactions.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mm-mind2web:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mm-mind2web", "languages": [], "modality": "multimodal", "name": "MM-Mind2Web", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.637, "raw_min": 0.558, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mm-mind2web:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500"}, "unit": null}, "slug": "llm-stats-mm-mind2web", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "communication"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multi-turn LLM-as-a-judge evaluation benchmark for testing multimodal instruction-tuned models' ability to follow user instructions in multi-turn dialogues and answer open-ended questions in a zero-shot manner.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mm-mt-bench:pixtral-12b-2409", "reported_at": "2024-09-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mm-mt-bench", "languages": [], "modality": "multimodal", "name": "MM-MT-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.9, "display_multiplier": 1, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 84.9, "raw_min": 0.06, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mm-mt-bench:mistral-large-3-2509", "reported_date": "2025-09-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-mm-mt-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "VQA", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MM-RLHF, a comprehensive project for aligning Multimodal Large Language Models (MLLMs) with human preferences. The dataset and algorithms enable consistent performance improvements across 10 dimensions and 27 benchmarks for open-source MLLMs. MM-RLHF，这是一个将多模态大型语言模型（MLLMs）与人类偏好对齐的全面项目，使开源多语言机器学习模型在 10 个维度和 27 个基准测试中实现持续的性能提升。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1539", "languages": [], "modality": "multimodal", "name": "MM-RLHF", "openness": "unknown", "publisher": "KuaiShou, CASIA, NJU", "released": "2025-02-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1539-mm-rlhf", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-RLHF", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MM-Vet evaluates the capabilities to deal with complicated multimodal tasks, defining 6 core VL capabilities and examining the 16 integrations of interest derived from the capability combination. MM-Vet用于评估复杂多模态任务能力，涵盖了6个核心视觉语言功能的16种功能组合。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1356", "languages": [], "modality": "multimodal", "name": "MM-Vet", "openness": "unknown", "publisher": "National University of Singapore", "released": "2023-08-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1356-mm-vet", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-Vet", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "VQA", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMAD is the first-ever full-spectrum MLLMs benchmark in industrial Anomaly Detection. \nResearchers defined seven key subtasks of MLLMs in industrial inspection and designed a novel pipeline to generate the MMAD dataset with 39,672 questions for 8,366 industrial images. MMAD是第一个工业异常检测领域的全谱 MLLMs 基准，研究人员定义了工业检测中 MLLMs 的七个关键子任务，并设计了一个新颖的流程来生成包含 39,672 个问题以及 8,366 个工业图像的 MMAD 数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1585", "languages": [], "modality": "multimodal", "name": "MMAD", "openness": "restricted", "publisher": "Southern University of Science and Technology, etc.", "released": "2024-10-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1585-mmad", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMAD", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "多模态", "Multimodal", "推理", "Reasoning", "MMAR", "音频语言模型", "深度推理", "多模态模型", "VLM", "逻辑推理", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "We introduce MMAR, a new benchmark comprising 1,000 meticulously curated audio-question-answer triplets, designed to evaluate the deep reasoning capabilities of Audio-Language Models (ALMs) across massive multi-disciplinary tasks. MMAR是一个全新评测基准，旨在评估音频-语言模型（ALMs）的深度推理能力。该基准包含1,000个精心构建的音频与问答，并经过多轮纠错与质量校验以确保高标准。与现有局限于特定声音、音乐或语音领域的评测体系不同，MMAR覆盖现实场景中的混合模态，并采用四级分层分类体系（信号层、感知层、语义层与文化层）。该基准中的每个题目都需要超越表层理解的多层次深度推理，部分问题更要求研究生级别的专业领域知识与感知能力。我们在MMAR上评估了多类模型，测试结果表明该基准具有显著挑战性，分析结果进一步揭示了当前模型在理解与推理能力上的关键局限。我们期待MMAR能推动这个重要但尚未充分探索的研究领域的发展。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1910", "languages": [], "modality": "multimodal", "name": "MMAR", "openness": "restricted", "publisher": "上海交通大学", "released": "2025-05-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1910-mmar", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMAR", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A massive multi-task audio understanding and reasoning benchmark comprising 10,000 carefully curated audio clips paired with human-annotated natural language questions spanning speech, environmental sounds, and music. Requires expert-level knowledge and complex reasoning across 27 distinct skills.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmau:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmau", "languages": [], "modality": "multimodal", "name": "MMAU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.77, "raw_min": 0.656, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmau:inkling-small", "reported_date": "2026-07-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500"}, "unit": null}, "slug": "llm-stats-mmau", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A subset of the MMAU benchmark focused specifically on music understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across music audio clips.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmau-music:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmau-music", "languages": [], "modality": "multimodal", "name": "MMAU Music", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 69.16, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.6916, "raw_min": 0.6916, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmau-music:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500"}, "unit": null}, "slug": "llm-stats-mmau-music", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A subset of the MMAU benchmark focused specifically on environmental sound understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across environmental sound clips.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmau-sound:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmau-sound", "languages": [], "modality": "multimodal", "name": "MMAU Sound", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.86999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.6787, "raw_min": 0.6787, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmau-sound:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500"}, "unit": null}, "slug": "llm-stats-mmau-sound", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "speech_to_text", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A subset of the MMAU benchmark focused specifically on speech understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across speech audio clips.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmau-speech:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmau-speech", "languages": [], "modality": "multimodal", "name": "MMAU Speech", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.760000000000005, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.5976, "raw_min": 0.5976, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmau-speech:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500"}, "unit": null}, "slug": "llm-stats-mmau-speech", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "knowledge", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMBC is a multimodal benchmark for vision-language knowledge and reasoning.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmbc:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmbc", "languages": [], "modality": "multimodal", "name": "MMBC", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 46.300000000000004, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.463, "raw_min": 0.463, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmbc:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500"}, "unit": null}, "slug": "llm-stats-mmbc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks with robust metrics.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmbench:phi-3.5-vision-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmbench", "languages": [], "modality": "multimodal", "name": "MMBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.8, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.918, "raw_min": 0.692, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmbench:step3-vl-10b", "reported_date": "2026-01-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500"}, "unit": null}, "slug": "llm-stats-mmbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "理解", "Understanding", "多模态模型", "VLM", "逻辑推理", "图像理解", "Image Understanding", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMBench is a collection of benchmarks to evaluate the multi-modal understanding capability of large vision language models (LVLMs). This benchmark contains 3,000 multiple-choice questions covering 20 fine-grained assessment dimensions. MMBench是OpenCompass 研究团队自建的视觉语言模型评测数据集，可实现从感知到认知能力逐级细分评估。此评测基准包含3000 道单项选择题 ，覆盖 20个细粒度评估维度。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1206", "languages": ["Chinese"], "modality": "multimodal", "name": "MMBench", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2023-07-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1206-mmbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Version 1.1 of MMBench, an improved bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks.", "evidence_summary": {"document_count": 1, "model_count": 20, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmbench-v1.1:deepseek-vl2", "reported_at": "2024-12-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:mmbench-v1.1", "languages": [], "modality": "multimodal", "name": "MMBench-V1.1", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 20, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.80000000000001, "display_multiplier": 100, "model_count": 20, "model_count_basis": "source_model_id", "numeric_count": 20, "raw_max": 0.928, "raw_min": 0.683, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmbench-v1.1:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500"}, "unit": null}, "slug": "llm-stats-mmbench-v1-1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A long-form multi-shot benchmark for holistic video understanding that incorporates approximately 600 web videos from YouTube spanning 16 major categories, with each video ranging from 30 seconds to 6 minutes. Includes roughly 2,000 original question-answer pairs covering 26 fine-grained capabilities.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmbench-video:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmbench-video", "languages": [], "modality": "multimodal", "name": "MMBench-Video", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 2.02, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.0202, "raw_min": 0.0179, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmbench-video:qwen2.5-vl-72b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500"}, "unit": null}, "slug": "llm-stats-mmbench-video", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "理解", "Understanding", "多模态模型", "VLM", "逻辑推理", "视频理解", "Video Understanding", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMBench-Video is a comprehensive video understanding evaluation benchmark that covers long videos, multiple shots, and evaluates the temporal understanding ability of MLLMs. Contains over 600 videos, 16 categories, and manually annotated Q&A pairs. MMBench-Video是全面视频理解评测基准，覆盖长视频、多镜头，评估MLLMs时序理解能力。包含16类共600+视频以及人工标注问答对。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1207", "languages": [], "modality": "multimodal", "name": "MMBench-Video", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-06-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1207-mmbench-video", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMBench-Video", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "NeurIPS 2024", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMDU is intended for evaluating the multi-image multi-turn dialogue capabilities. It comprises 110 high-quality multi-image multi-turn dialogues with more than 1600 questions, each accompanied by detailed long-form answers. MMDU用于评估大型视觉语言模型的多图像多轮对话能力，包含110个高质量的多图像多轮对话，由1600多个附有详细的长篇答案的问题组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1334", "languages": [], "modality": "multimodal", "name": "MMDU", "openness": "restricted", "publisher": "Shanghai Jiao Tong University", "released": "2024-06-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1334-mmdu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMDU", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive evaluation benchmark for Multimodal Large Language Models measuring both perception and cognition abilities across 14 subtasks. Features manually designed instruction-answer pairs to avoid data leakage and provides systematic quantitative assessment of MLLM capabilities.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mme:deepseek-vl2", "reported_at": "2024-12-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mme", "languages": [], "modality": "multimodal", "name": "MME", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.1, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.731, "raw_min": 0.1915, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mme:lfm-2.5-vl-3b", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500"}, "unit": null}, "slug": "llm-stats-mme", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MME is a comprehensive MLLM evaluation benchmark. It measures both perception and cognition abilities on a total of 14 subtasks. MME是一个全面的多模态大模型评估基准，涵盖14 个考察感知和认知能力的子任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1357", "languages": [], "modality": "multimodal", "name": "MME", "openness": "open", "publisher": "Tencent Youtu Lab", "released": "2023-06-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1357-mme", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MME", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "多模态模型", "VLM", "逻辑推理", "Reasoning", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MME-CoT is a specialized benchmark evaluating the CoT reasoning performance of LMMs, spanning six domains: math, science, OCR, logic, space-time, and general scenes. MME-CoT，一个专门用于评估 LMMs CoT 推理性能的基准，涵盖六个领域：数学、科学、OCR、逻辑、时空和一般场景。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1565", "languages": [], "modality": null, "name": "MME-CoT", "openness": "restricted", "publisher": "CUHK MMLab, CUHK MiuLar Lab, ByteDance, etc.", "released": "2025-02-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1565-mme-cot", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MME-CoT", "unit": null}, {"aliases": [], "categories": ["multimodal", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive evaluation benchmark for Multimodal Large Language Models featuring over 13,366 high-resolution images and 29,429 question-answer pairs across 43 subtasks and 5 real-world scenarios. The largest manually annotated multimodal benchmark to date, designed to test MLLMs on challenging high-resolution real-world scenarios.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mme-realworld:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mme-realworld", "languages": [], "modality": "multimodal", "name": "MME-RealWorld", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.616, "raw_min": 0.616, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mme-realworld:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500"}, "unit": null}, "slug": "llm-stats-mme-realworld", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MME-RealWorld evaluates MLLMs' real-world recognition, featuring 13,366 high-resolution images averaging 2,000 × 1,500 pixels. MME-RealWorld用于评估多模态大模型对真实场景的理解能力，包含13366个平均2000*1500像素的高分辨率图像。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1451", "languages": ["English", "Chinese"], "modality": "multimodal", "name": "MME-RealWorld", "openness": "unknown", "publisher": "Chinese Academy of Sciences", "released": "2024-08-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1451-mme-realworld", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MME-RealWorld", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMIE, a large-scale knowledge-intensive benchmark for evaluating interleaved multimodal comprehension and generation in Large Vision-Language Models (LVLMs). MMIE，这是一个大规模知识密集型基准，用于评估大型视觉-语言模型（LVLMs）中的交错多模态理解和生成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1523", "languages": [], "modality": "multimodal", "name": "MMIE", "openness": "restricted", "publisher": "UNC-Chapel Hill; University of Chicago; Microsoft Research; NUS", "released": "2024-10-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1523-mmie", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMIE", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A benchmark for evaluating Multimodal Large Language Models (MLLMs) on detecting and reasoning about inconsistencies in layout-rich multimodal content. MMIR features 534 challenging samples across five reasoning-heavy inconsistency categories. 用于评估多模态模型（MLLM）检测和推理布局丰富的多模态内容中的不一致性的基准。MMIR 包含 534 个具有挑战性的样本，涉及五个推理能力较强的不一致类别。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1547", "languages": [], "modality": "multimodal", "name": "MMIR", "openness": "unknown", "publisher": "University of California,  eBay", "released": "2025-02-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1547-mmir", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMIR", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMIU test MLLMs' multi-image understanding capabilities, encompassesing 7 types of multi-image relationships, 52 tasks, 77K images, and 11K meticulously curated multiple-choice questions. MMIU用于评估多模态大模型的多图理解能力，包含7种类型的多图像关系、52个任务、77K图像和11K精心策划的多选题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1453", "languages": [], "modality": "multimodal", "name": "MMIU", "openness": "restricted", "publisher": "Shanghai AI Laboratory", "released": "2024-08-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1453-mmiu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMIU", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMKE-Bench is a benchmark designed to evaluate the ability of LMMs to edit visual knowledge in real-world scenarios. It includes 2,940 pieces of knowledge and 8,363 images across 33 broad categories, with automatically generated, human-verified evaluation questions. MMKE-Bench是一个旨在评估 LMM 在现实场景中编辑视觉知识能力的基准，包括 33 个广泛类别中的 2,940 条知识和 8,363 张图像，以及自动生成并由人工验证的评估问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1578", "languages": [], "modality": null, "name": "MMKE-Bench", "openness": "restricted", "publisher": "BIGAI, USTC, PKU, BIT", "released": "2025-02-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1578-mmke-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMKE-Bench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "知识", "Knowledge", "长文本", "Long-Context", "Vision-Language", "多模态模型", "VLM", "知识储备", "长上下文", "Long Context", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMLongBench is a benchmark that evaluates long-context vision-language models across various tasks, image types, and input lengths, revealing that single-task performance is insufficient for gauging overall vision-language long-context capability. MMLongBench 是一个针对长上下文视觉-语言模型的基准，覆盖多种任务、图像类型和输入长度。评测结果表明，单一任务的表现不足以衡量模型的整体长上下文能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1851", "languages": [], "modality": "multimodal", "name": "MMLongBench", "openness": "open", "publisher": "HKUST, Tencent AI Seattle Lab, University of Edinburgh, Miniml.AI, NVIDIA AI", "released": "2025-05-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1851-mmlongbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench", "unit": null}, {"aliases": [], "categories": ["long_context", "multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMLongBench-128K evaluates multimodal long-context understanding at a 128K token context length, testing how well vision-language models reason over very long mixed text and image inputs.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlongbench-128k:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlongbench-128k", "languages": [], "modality": "multimodal", "name": "MMLongBench-128K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.3, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.783, "raw_min": 0.769, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlongbench-128k:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlongbench-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMLongBench-Doc evaluates long document understanding capabilities in vision-language models.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlongbench-doc:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlongbench-doc", "languages": [], "modality": "image", "name": "MMLongBench-Doc", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.62, "display_multiplier": 1, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.62, "raw_min": 0.562, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlongbench-doc:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlongbench-doc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "NeurIPS 2024", "多模态模型", "VLM", "长上下文", "Long Context", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMLongBench-Doc is a long-context multi-modal benchmark comprising 1,062 expert-annotated questions. MMLONGBENCH-DOC是一个长上下文的多模态基准，由1062个专家注释的问题组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1275", "languages": [], "modality": null, "name": "MMLongBench-Doc", "openness": "unknown", "publisher": "Nanyang Technological University", "released": "2024-07-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1275-mmlongbench-doc", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench-Doc", "unit": null}, {"aliases": [], "categories": ["legal", "math", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Massive Multitask Language Understanding benchmark testing knowledge across 57 diverse subjects including STEM, humanities, social sciences, and professional domains", "evidence_summary": {"document_count": 1, "model_count": 101, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:mmlu", "languages": [], "modality": "text", "name": "MMLU", "openness": "unknown", "publisher": null, "released": "2020-09-07", "released_reference": {"basis": "paper_first_version", "note": "First version introducing MMLU; this is a benchmark date, not a recent model release.", "source_key": "opencompass:498", "source_url": "https://arxiv.org/abs/2009.03300"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 101, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.5, "display_multiplier": 100, "model_count": 101, "model_count_basis": "source_model_id", "numeric_count": 101, "raw_max": 0.925, "raw_min": 0.419, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu:gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500", "unit": null}, {"aliases": ["MMLU"], "categories": ["knowledge"], "collected_at": null, "description": "Saturated at the frontier and heavily contaminated. Useful as a historical baseline, not as a ranking signal.", "evidence_summary": {"document_count": 9, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mmlu\u0000google_gemini_1_5_report\u0000mmlu\u00005-shot\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "first_score_reported_at": "2024-03-08", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mmlu\u0000google_gemini_1_5_report\u0000mmlu\u00005-shot\u0000Gemini 1.5 Pro", "reported_at": "2024-03-08", "source_url": "https://arxiv.org/abs/2403.05530"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:mmlu", "languages": [], "modality": null, "name": "MMLU", "openness": "unknown", "publisher": null, "released": "2020-09-07", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mmlu", "source_url": "https://github.com/hendrycks/test"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.8, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 90.8, "raw_min": 84.0, "source_reference": {"obs_id": "curated\u0000mmlu\u0000deepseek_r1_report\u0000mmlu\u0000EM\u0000DeepSeek-R1", "observation_id": "curated\u0000mmlu\u0000deepseek_r1_report\u0000mmlu\u0000EM\u0000DeepSeek-R1", "reported_at": "2025-01-22", "reported_date": "2025-01-22", "source_id": "deepseek_r1_report", "source_url": "https://arxiv.org/abs/2501.12948"}, "unit": "percent"}, "slug": "mmlu", "source": "model_reports", "source_url": "https://github.com/hendrycks/test", "unit": "percent"}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMLU (Massive Multitask Language Understanding) is a new benchmark designed to measure knowledge acquired during pretraining by evaluating models exclusively in zero-shot and few-shot settings. This makes the benchmark more challenging and more similar to how we evaluate humans. The benchmark covers 57 subjects across STEM, the humanities, the social sciences, and more. It ranges in difficulty from an elementary level to an advanced professional level, and it tests both world knowledge and problem solving ability. Subjects range from traditional areas, such as mathematics and history, to more specialized areas like law and ethics. The granularity and breadth of the subjects makes the benchmark ideal for identifying a model’s blind spots. MMLU (Massive Multitask Language Understanding) 是一个新的基准测试，旨在通过在零次学习和少次学习的环境中评估模型来测量预训练期间获得的知识。这使得基准测试更具挑战性，且更接近我们评估人类的方式。该基准测试涵盖了STEM、人文学科、社会科学等57个主题。其难度范围从小学级别到专业级别，旨在测试世界知识和解决问题的能力。测试主题范围从传统领域，如数学和历史，到更专业的领域，如法律和伦理学。题目的精细度和广度使该基准测试成为识别模型盲点的理想选择。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:498", "languages": [], "modality": null, "name": "MMLU", "openness": "unknown", "publisher": null, "released": "2020-09-07", "released_reference": {"basis": "paper_first_version", "note": "First version introducing MMLU; this is a benchmark date, not a recent model release.", "source_key": "opencompass:498", "source_url": "https://arxiv.org/abs/2009.03300"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-498-mmlu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLU", "unit": null}, {"aliases": [], "categories": ["legal", "math", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Chain-of-Thought variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses chain-of-thought prompting to elicit step-by-step reasoning.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-(cot):llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlu-(cot)", "languages": [], "modality": "text", "name": "MMLU (CoT)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.6, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.886, "raw_min": 0.73, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-(cot):llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-cot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "math", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Chat-format variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses conversational prompting format for model evaluation.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-chat:llama-3.1-nemotron-70b-instruct", "reported_at": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlu-chat", "languages": [], "modality": "text", "name": "MMLU Chat", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.58, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.8058, "raw_min": 0.8058, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-chat:llama-3.1-nemotron-70b-instruct", "reported_date": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-chat", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "math", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "French language variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This multilingual version tests model performance in French.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-french:mistral-large-2-2407", "reported_at": "2024-07-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlu-french", "languages": [], "modality": "text", "name": "MMLU French", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.828, "raw_min": 0.828, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-french:mistral-large-2-2407", "reported_date": "2024-07-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-french", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "math", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Base version of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. Designed to comprehensively measure the breadth and depth of a model's academic and professional understanding.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-base:qwen-2.5-coder-7b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlu-base", "languages": [], "modality": "text", "name": "MMLU-Base", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.68, "raw_min": 0.68, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-base:qwen-2.5-coder-7b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-base", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "math", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A more robust and challenging multi-task language understanding benchmark that extends MMLU by expanding multiple-choice options from 4 to 10, eliminating trivial questions, and focusing on reasoning-intensive tasks. Features over 12,000 curated questions across 14 domains and causes a 16-33% accuracy drop compared to original MMLU.", "evidence_summary": {"document_count": 1, "model_count": 134, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-pro:claude-3-opus-20240229", "reported_at": "2024-02-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mmlu-pro", "languages": [], "modality": "text", "name": "MMLU-Pro", "openness": "open", "publisher": "University of Waterloo", "released": "2024-06-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 134, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.33, "display_multiplier": 100, "model_count": 134, "model_count_basis": "source_model_id", "numeric_count": 134, "raw_max": 0.9033, "raw_min": 0.147, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-pro:sakana-namazu", "reported_date": "2026-08-03", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500", "unit": null}, {"aliases": ["MMLU-Pro", "MMLU Pro"], "categories": ["knowledge"], "collected_at": null, "description": "Static closed-set multiple choice; contamination risk rises over time.", "evidence_summary": {"document_count": 12, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mmlu_pro\u0000deepseek_v3_report\u0000mmlu_pro\u0000EM, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mmlu_pro\u0000deepseek_v3_report\u0000mmlu_pro\u0000EM, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:mmlu_pro", "languages": [], "modality": null, "name": "MMLU-Pro", "openness": "unknown", "publisher": null, "released": "2024-06-03", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mmlu_pro", "source_url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.8, "display_multiplier": 1, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 87.8, "raw_min": 75.9, "source_reference": {"obs_id": "curated\u0000mmlu_pro\u0000qwen3_5_model_card\u0000mmlu_pro\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000mmlu_pro\u0000qwen3_5_model_card\u0000mmlu_pro\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "mmlu_pro", "source": "model_reports", "source_url": "https://github.com/TIGER-AI-Lab/MMLU-Pro", "unit": "percent"}, {"aliases": [], "categories": ["学科", "Examination", "NeurIPS 2024", "大语言模型", "LLM", "知识储备", "Knowledge", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMLU-Pro is the extension of MMLU, integrating more challenging, reasoning-focused questions and expanding the choice set from four to ten options. MMLU-Pro是MMLU的扩展版本，涵盖了更具挑战性、以推理为重点的问题，并将选择集从4个选项扩展到10个选项。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1276", "languages": [], "modality": null, "name": "MMLU-Pro", "openness": "open", "publisher": "University of Waterloo", "released": "2024-06-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1276-mmlu-pro", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLU-Pro", "unit": null}, {"aliases": [], "categories": ["legal", "math", "reasoning", "language", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Extended version of MMLU-Pro providing additional challenging multiple-choice questions for evaluating language models across diverse academic and professional domains. Built on the foundation of the Massive Multitask Language Understanding benchmark framework.", "evidence_summary": {"document_count": 1, "model_count": 32, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-prox:gemma-3n-e2b-it-litert-preview", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:mmlu-prox", "languages": [], "modality": "text", "name": "MMLU-ProX", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 32, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.0, "display_multiplier": 100, "model_count": 32, "model_count_basis": "source_model_id", "numeric_count": 32, "raw_max": 0.87, "raw_min": 0.081, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-prox:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-prox", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "An improved version of the MMLU benchmark featuring manually re-annotated questions to identify and correct errors in the original dataset. Provides more reliable evaluation metrics for language models by addressing dataset quality issues found in the original MMLU.", "evidence_summary": {"document_count": 1, "model_count": 48, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-redux:qwen-2.5-14b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mmlu-redux", "languages": [], "modality": "text", "name": "MMLU-Redux", "openness": "open", "publisher": null, "released": "2024-06-06", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 48, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.0, "display_multiplier": 100, "model_count": 48, "model_count_basis": "source_model_id", "numeric_count": 48, "raw_max": 0.95, "raw_min": 0.432, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-redux:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-redux", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500", "unit": null}, {"aliases": ["MMLU-Redux"], "categories": ["knowledge"], "collected_at": null, "description": "Error-corrected MMLU subset; reported mainly by open-weight cards.", "evidence_summary": {"document_count": 5, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mmlu_redux\u0000deepseek_v3_report\u0000mmlu_redux\u0000EM, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mmlu_redux\u0000deepseek_v3_report\u0000mmlu_redux\u0000EM, 8K output limit\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:mmlu_redux", "languages": [], "modality": null, "name": "MMLU-Redux", "openness": "unknown", "publisher": null, "released": "2024-06-06", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mmlu_redux", "source_url": "https://github.com/aryopg/mmlu-redux"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.9, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 94.9, "raw_min": 89.1, "source_reference": {"obs_id": "curated\u0000mmlu_redux\u0000qwen3_5_model_card\u0000mmlu_redux\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000mmlu_redux\u0000qwen3_5_model_card\u0000mmlu_redux\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "mmlu_redux", "source": "model_reports", "source_url": "https://github.com/aryopg/mmlu-redux", "unit": "percent"}, {"aliases": [], "categories": ["math", "reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A curated version of the MMLU benchmark featuring manually re-annotated 5,700 questions across 57 subjects to identify and correct errors in the original dataset. Addresses the 6.49% error rate found in MMLU and provides more reliable evaluation metrics for language models.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-redux-2.0:kimi-k2-base", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlu-redux-2.0", "languages": [], "modality": "text", "name": "MMLU-redux-2.0", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.2, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.902, "raw_min": 0.902, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-redux-2.0:kimi-k2-base", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-redux-2-0", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "physics", "reasoning", "chemistry"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "STEM-focused subset of the Massive Multitask Language Understanding benchmark, evaluating language models on science, technology, engineering, and mathematics topics including physics, chemistry, mathematics, and other technical subjects.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmlu-stem:qwen-2.5-14b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmlu-stem", "languages": [], "modality": "text", "name": "MMLU-STEM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.9, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.809, "raw_min": 0.764, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmlu-stem:qwen-2.5-32b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500"}, "unit": null}, "slug": "llm-stats-mmlu-stem", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Multilingual Massive Multitask Language Understanding dataset released by OpenAI, featuring professionally translated MMLU test questions across 14 languages including Arabic, Bengali, German, Spanish, French, Hindi, Indonesian, Italian, Japanese, Korean, Portuguese, Swahili, Yoruba, and Chinese. Contains approximately 15,908 multiple-choice questions per language covering 57 subjects.", "evidence_summary": {"document_count": 1, "model_count": 49, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmmlu:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "llm-stats:mmmlu", "languages": [], "modality": "text", "name": "MMMLU", "openness": "open", "publisher": "OpenAI", "released": null, "released_reference": null, "repo_kind": "harness_only", "repo_resolution_status": "resolved", "score_count": 49, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.7, "display_multiplier": 100, "model_count": 49, "model_count_basis": "source_model_id", "numeric_count": 49, "raw_max": 0.927, "raw_min": 0.443, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmmlu:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500"}, "unit": null}, "slug": "llm-stats-mmmlu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500", "unit": null}, {"aliases": ["MMMLU", "Multilingual MMLU", "MMLU-ProX"], "categories": ["multilingual"], "collected_at": null, "description": "A language-count average. Two cards reporting \"MMMLU\" may be averaging over different language sets, so the counts must match to compare.", "evidence_summary": {"document_count": 7, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mmmlu\u0000anthropic_claude_4_system_card\u0000mmmlu_14_non_english_languages\u000014 non-English languages, extended thinking up to 64K tokens\u0000Claude Opus 4", "reported_at": "2025-05-22", "source_url": "https://www.anthropic.com/news/claude-4"}, "first_score_reported_at": "2025-05-22", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mmmlu\u0000anthropic_claude_4_system_card\u0000mmmlu_14_non_english_languages\u000014 non-English languages, extended thinking up to 64K tokens\u0000Claude Opus 4", "reported_at": "2025-05-22", "source_url": "https://www.anthropic.com/news/claude-4"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mmmlu", "languages": [], "modality": null, "name": "MMMLU", "openness": "unknown", "publisher": null, "released": "2024-09-24", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mmmlu", "source_url": "https://huggingface.co/datasets/openai/MMMLU"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.6, "display_multiplier": 1, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 92.6, "raw_min": 86.5, "source_reference": {"obs_id": "curated\u0000mmmlu\u0000google_gemini_3_1_pro_model_card\u0000mmmlu\u0000Thinking (High)\u0000Gemini 3.1 Pro", "observation_id": "curated\u0000mmmlu\u0000google_gemini_3_1_pro_model_card\u0000mmmlu\u0000Thinking (High)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "reported_date": "2026-02-19", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "unit": "percent"}, "slug": "mmmlu", "source": "model_reports", "source_url": "https://huggingface.co/datasets/openai/MMMLU", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMMU (Massive Multi-discipline Multimodal Understanding) is a benchmark designed to evaluate multimodal models on college-level subject knowledge and deliberate reasoning. Contains 11.5K meticulously collected multimodal questions from college exams, quizzes, and textbooks, covering six core disciplines: Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, and Tech & Engineering across 30 subjects and 183 subfields.", "evidence_summary": {"document_count": 1, "model_count": 63, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmmu:gpt-3.5-turbo-0125", "reported_at": "2023-03-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mmmu", "languages": [], "modality": "multimodal", "name": "MMMU", "openness": "open", "publisher": "IN.AI Research", "released": "2023-11-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 63, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.0, "display_multiplier": 100, "model_count": 63, "model_count_basis": "source_model_id", "numeric_count": 63, "raw_max": 0.86, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmmu:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500"}, "unit": null}, "slug": "llm-stats-mmmu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500", "unit": null}, {"aliases": ["MMMU", "MMMU-Pro"], "categories": ["multimodal"], "collected_at": null, "description": "Image preprocessing and chain-of-thought settings affect results.", "evidence_summary": {"document_count": 10, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mmmu\u0000anthropic_claude_4_system_card\u0000mmmu\u0000extended thinking up to 64K tokens\u0000Claude Opus 4", "reported_at": "2025-05-22", "source_url": "https://www.anthropic.com/news/claude-4"}, "first_score_reported_at": "2025-05-22", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mmmu\u0000anthropic_claude_4_system_card\u0000mmmu\u0000extended thinking up to 64K tokens\u0000Claude Opus 4", "reported_at": "2025-05-22", "source_url": "https://www.anthropic.com/news/claude-4"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mmmu", "languages": [], "modality": null, "name": "MMMU", "openness": "unknown", "publisher": null, "released": "2023-11-27", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mmmu", "source_url": "https://mmmu-benchmark.github.io/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.0, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 85.0, "raw_min": 74.4, "source_reference": {"obs_id": "curated\u0000mmmu\u0000qwen3_5_model_card\u0000mmmu\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000mmmu\u0000qwen3_5_model_card\u0000mmmu\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "mmmu", "source": "model_reports", "source_url": "https://mmmu-benchmark.github.io/", "unit": "percent"}, {"aliases": [], "categories": ["知识", "Knowledge", "Multimodal", "多模态模型", "VLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMMU is a new benchmark designed to evaluate multimodal models on massive multi-discipline tasks demanding college-level subject knowledge and deliberate reasoning. MMMU includes 11.5K meticulously collected multimodal questions from college exams, quizzes, and textbooks. MMMU用于评估多模态大模型在复杂多学科任务中的表现，包括从大学考试和教科书中精心收集的11.5K多模态问题，涵盖六个核心学科：艺术与设计、商业、科学、健康与医学、人文与社会科学以及技术与工程。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1248", "languages": [], "modality": null, "name": "MMMU", "openness": "open", "publisher": "IN.AI Research", "released": "2023-11-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1248-mmmu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMMU", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Validation set of the Massive Multi-discipline Multimodal Understanding and Reasoning benchmark. Features college-level multimodal questions across 6 core disciplines (Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, Tech & Engineering) spanning 30 subjects and 183 subfields with diverse image types including charts, diagrams, maps, and tables.", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmmu-(val):gemma-3-12b-it", "reported_at": "2025-03-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmmu-(val)", "languages": [], "modality": "multimodal", "name": "MMMU (val)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.10000000000001, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.781, "raw_min": 0.329, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmmu-(val):qwen3-vl-32b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mmmu-val", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Validation set of the Massive Multi-discipline Multimodal Understanding and Reasoning benchmark. Features college-level multimodal questions across 6 core disciplines (Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, Tech & Engineering) spanning 30 subjects and 183 subfields with diverse image types including charts, diagrams, maps, and tables.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmmu-(validation):claude-opus-4-20250514", "reported_at": "2025-05-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmmu-(validation)", "languages": [], "modality": "multimodal", "name": "MMMU (validation)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.7, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.807, "raw_min": 0.732, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmmu-(validation):claude-opus-4-5-20251101", "reported_date": "2025-11-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mmmu-validation", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning"], "collected_at": "2026-08-25T10:41:06Z", "description": "Visual reasoning", "evidence_summary": {"document_count": 1, "model_count": 245, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:mmmu-pro:1fc54cef-d179-48b1-a27d-046874e9b208", "reported_at": "2024-03-04", "source_url": "https://artificialanalysis.ai/evaluations/mmmu-pro"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:mmmu-pro", "languages": [], "modality": null, "name": "MMMU-Pro", "openness": "unknown", "publisher": null, "released": "2024-09-05", "released_reference": {"basis": "release_announcement", "note": "The benchmark authors announce MMMU-Pro on September 5. The earlier MMMU release and the later public MMMU test answers do not date this distinct benchmark.", "source_key": "artificial-analysis:mmmu-pro", "source_url": "https://github.com/MMMU-Benchmark/MMMU"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 245, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.4913294797688, "display_multiplier": 100, "model_count": 245, "model_count_basis": "source_model_id", "numeric_count": 245, "raw_max": 0.854913294797688, "raw_min": 0.145086705202312, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:mmmu-pro:b2331108-72ed-415a-82d1-188633875bbc", "reported_date": "2026-08-13", "source_url": "https://artificialanalysis.ai/evaluations/mmmu-pro"}, "unit": null}, "slug": "artificial-analysis-mmmu-pro", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/mmmu-pro", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A more robust multi-discipline multimodal understanding benchmark that enhances MMMU through a three-step process: filtering text-only answerable questions, augmenting candidate options, and introducing vision-only input settings. Achieves significantly lower model performance (16.8-26.9%) compared to original MMMU, providing more rigorous evaluation that closely mimics real-world scenarios.", "evidence_summary": {"document_count": 1, "model_count": 68, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmmu-pro:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mmmu-pro", "languages": [], "modality": "multimodal", "name": "MMMU-Pro", "openness": "open", "publisher": null, "released": "2024-09-04", "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 68, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.6, "display_multiplier": 100, "model_count": 68, "model_count_basis": "source_model_id", "numeric_count": 68, "raw_max": 0.836, "raw_min": 0.305, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmmu-pro:gemini-3.5-flash", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-mmmu-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500", "unit": null}, {"aliases": ["MMMU-Pro", "MMMU Pro"], "categories": ["multimodal"], "collected_at": null, "description": "Harder vision-required subset of MMMU. Input image ordering changes the result, so the protocol has to travel with the number.", "evidence_summary": {"document_count": 4, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mmmu_pro\u0000qwen3_5_model_card\u0000mmmu_pro\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mmmu_pro\u0000qwen3_5_model_card\u0000mmmu_pro\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:mmmu_pro", "languages": [], "modality": null, "name": "MMMU-Pro", "openness": "unknown", "publisher": null, "released": "2024-09-04", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mmmu_pro", "source_url": "https://arxiv.org/abs/2409.02813"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.4, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 83.4, "raw_min": 76.9, "source_reference": {"obs_id": "curated\u0000mmmu_pro\u0000moonshot_kimi_k3_model_card\u0000mmmu_pro\u0000reasoning=max, with Python\u0000Kimi K3", "observation_id": "curated\u0000mmmu_pro\u0000moonshot_kimi_k3_model_card\u0000mmmu_pro\u0000reasoning=max, with Python\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "mmmu_pro", "source": "model_reports", "source_url": "https://arxiv.org/abs/2409.02813", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMMU-Pro variant evaluated with tool access enabled.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmmu-pro-with-tools:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmmu-pro-with-tools", "languages": [], "modality": "multimodal", "name": "MMMU-Pro (with tools)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.6, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.846, "raw_min": 0.795, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmmu-pro-with-tools:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500"}, "unit": null}, "slug": "llm-stats-mmmu-pro-with-tools", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Validation set for MMMU (Massive Multi-discipline Multimodal Understanding and Reasoning) benchmark, designed to evaluate multimodal models on massive multi-discipline tasks demanding college-level subject knowledge and deliberate reasoning across Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, and Tech & Engineering.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmmuval:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmmuval", "languages": [], "modality": "multimodal", "name": "MMMUval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.60000000000001, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.806, "raw_min": 0.645, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmmuval:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500"}, "unit": null}, "slug": "llm-stats-mmmuval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "语言", "Language", "Visual Place Recognition", "Dataset and Benchmark", "多模态模型", "VLM", "图像理解", "Image Understanding", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMS-VPR is a large-scale multimodal dataset for street-level place recognition in pedestrian areas. It contains 78,575 images and 2,512 videos from 207 locations in Chengdu, China, with rich metadata and spatial graph structure. MMS-VPR 是一个面向复杂城市步行街区的 多模态街景视觉地点识别大规模数据集，采集自成都约 70,800 平方米的开放式商业街区，覆盖 207 个地点，包含 78,575 张图像和 2,512 段视频，每条数据均带有 GPS 坐标、时间戳和文本元信息。相较以车载视角和西方城市为主的传统数据集，MMS-VPR 更贴近真实、密集、多用途的街道空间，具备丰富的视角、时段和模态多样性。此外，该数据集构建了包含 81 个节点、125 条边的空间图结构，支持结构感知的地点识别方法。数据集还定义了两个子集（Edges 和Points），支持精细化和图结构评估任务，助力多模态与地理空间理解的交叉研究。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1829", "languages": [], "modality": "multimodal", "name": "MMS-VPR", "openness": "restricted", "publisher": "University of Auckland & Hunan University", "released": "2025-05-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1829-mms-vpr", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMS-VPR", "unit": null}, {"aliases": [], "categories": ["multimodal", "search", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMSearch evaluates multimodal models on search-based retrieval and question answering tasks that require processing both visual and textual information from search results.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmsearch:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmsearch", "languages": [], "modality": "multimodal", "name": "MMSearch", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.89999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.729, "raw_min": 0.729, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmsearch:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500"}, "unit": null}, "slug": "llm-stats-mmsearch", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "检索能力", "Retrieval", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Logo MMSearch is a multimodal search benchmark crafted to evaluate the potential of LMMs to function as a multimodal AI search engine. This benchmark encompasses a meticulously collected dataset of 300 queries spanning 14 subfields. MMSearch 是一个多模态搜索基准，旨在评估大型语言模型（LMMs）作为多模态 AI 搜索引擎的潜力。该基准包含了一个精心收集的包含 300 个查询的数据集，涵盖 14 个子领域。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1584", "languages": [], "modality": "multimodal", "name": "MMSearch", "openness": "unknown", "publisher": "CUHK MMLab, ByteDance, etc.", "released": "2024-09-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1584-mmsearch", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMSearch", "unit": null}, {"aliases": [], "categories": ["multimodal", "search", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMSearch-Plus is an extended variant of MMSearch with harder multimodal search and retrieval tasks requiring deeper reasoning over visual and textual search results.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmsearch-plus:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmsearch-plus", "languages": [], "modality": "multimodal", "name": "MMSearch-Plus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 41.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.414, "raw_min": 0.3, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmsearch-plus:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500"}, "unit": null}, "slug": "llm-stats-mmsearch-plus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "空间智能", "跨视角", "物理智能", "Embodied AI", "逻辑推理", "空间理解", "Spatial Understanding", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMSI-Bench is a novel Visual Question Answering (VQA) benchmark specifically designed to evaluate Multi-image Spatial Intelligence in multimodal large language models (MLLMs). Unlike traditional datasets that focus on spatial reasoning within a single image, MMSI-Bench emphasizes real-world inspired MMSI-Bench 是一个全新的多模态空间智能视觉问答（VQA）基准数据集，专为评估多图像空间推理能力而设计。与专注于单图像关系推理的传统数据集不同，MMSI-Bench 更贴近现实世界，聚焦于需在多张图像之间进行逻辑推理的复杂场景。该数据集由六位三维视觉专家耗时超过300小时构建，精心整理出包含1,000个高质量选择题的问题集，题目来自12万余张图像，并附有精心设计的误导选项与逐步推理过程。对34个主流开源和闭源多模态大语言模型的实证评估显示：最先进的模型准确率仅为30%至40%，而人类表现高达97%，揭示该任务的巨大挑战性与模型发展空间。此外，MMSI-Bench 配套提供自动化错误分析", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1922", "languages": [], "modality": "multimodal", "name": "MMSI-Bench", "openness": "unknown", "publisher": "上海人工智能实验室", "released": "2025-05-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1922-mmsi-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMSI-Bench", "unit": null}, {"aliases": [], "categories": ["multimodal", "spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMSIBench is a multimodal spatial-intelligence benchmark evaluating spatial reasoning over images.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmsibench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmsibench", "languages": [], "modality": "multimodal", "name": "MMSIBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 35.9, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.359, "raw_min": 0.314, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmsibench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500"}, "unit": null}, "slug": "llm-stats-mmsibench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMStar is an elite vision-indispensable multimodal benchmark comprising 1,500 challenge samples meticulously selected by humans to evaluate 6 core capabilities and 18 detailed axes. The benchmark addresses issues of visual content unnecessity and unintentional data leakage in existing multimodal evaluations.", "evidence_summary": {"document_count": 1, "model_count": 24, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmstar:deepseek-vl2", "reported_at": "2024-12-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:mmstar", "languages": [], "modality": "multimodal", "name": "MMStar", "openness": "restricted", "publisher": "Shanghai AI Laboratory", "released": "2024-04-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 24, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.3, "display_multiplier": 100, "model_count": 24, "model_count_basis": "source_model_id", "numeric_count": 24, "raw_max": 0.833, "raw_min": 0.459, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmstar:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500"}, "unit": null}, "slug": "llm-stats-mmstar", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMStar is an elite vision-indispensable multi-modal benchmark comprising 1,500 samples meticulously selected by humans. MMStar benchmarks 6 core capabilities and 18 detailed axes, aiming to evaluate LVLMs’ multi-modal capacities with carefully balanced and purified samples. MMStar 是一个多模态基准，包含 1,500 个经过人工精心挑选的样本。MMStar 评估 6 项核心能力和 18 个详细维度，旨在通过精心平衡和净化的样本，评估 LVLM 的多模态能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1175", "languages": [], "modality": null, "name": "MMStar", "openness": "restricted", "publisher": "Shanghai AI Laboratory", "released": "2024-04-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1175-mmstar", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMStar", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMT-Bench is a comprehensive multimodal benchmark for evaluating Large Vision-Language Models towards multitask AGI. It comprises 31,325 meticulously curated multi-choice visual questions from various multimodal scenarios such as vehicle driving and embodied navigation, covering 32 core meta-tasks and 162 subtasks in multimodal understanding.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmt-bench:deepseek-vl2", "reported_at": "2024-12-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmt-bench", "languages": [], "modality": "multimodal", "name": "MMT-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.6, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.636, "raw_min": 0.532, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmt-bench:deepseek-vl2", "reported_date": "2024-12-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-mmt-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MMT-Bench is designed to assess LVLMs across massive multimodal tasks requiring expert knowledge and deliberate visual recognition, localization, reasoning, and planning. It comprises 31,325 multi-choice visual questions, covering 32 core meta-tasks and 162 subtasks in multimodal understanding. MMT-Bench考察多模态大模型的视觉识别、定位、推理和规划能力，包括31325个多选视觉问题，涵盖了32个核心元任务和162个多模态理解子任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1364", "languages": [], "modality": "multimodal", "name": "MMT-Bench", "openness": "open", "publisher": "Shanghai AI Laboratory", "released": "2024-04-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1364-mmt-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMT-Bench", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Large language models (LLMs) demonstrate strong potential as agents for tool invocation due to their advanced comprehension and planning capabilities. Users increasingly rely on LLM-based agents to solve complex missions through iterative interactions. However, existing benchmarks predominantly acce 大型语言模型（LLM）凭借其先进的理解能力和规划能力，在作为工具调用的智能体方面展现出了巨大的潜力。用户越来越依赖基于大型语言模型的智能体，通过迭代交互来解决复杂任务。 然而，现有的基准测试主要是在单任务场景中评估智能体，无法体现现实世界的复杂性。为了填补这一空白，我们提出了 Multi-Mission Tool Bench 测试。在这个基准测试中，每个测试用例都包含多个相互关联的任务。这种设计要求智能体能够动态适应不断变化的需求。此外，所提出的基准测试探索了在固定任务数量下所有可能的任务切换模式。具体而言，我们提出了一个多智能体数据生成框架来构建这个基准测试。我们还提出了一种利用动态决策树来", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1736", "languages": ["English", "Chinese"], "modality": null, "name": "MMTB", "openness": "unknown", "publisher": "Tencent", "released": "2025-03-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1736-mmtb", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMTB", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "spatial_reasoning", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MM-Vet is an evaluation benchmark that examines large multimodal models on complicated multimodal tasks requiring integrated capabilities. It assesses six core vision-language capabilities: recognition, knowledge, spatial awareness, language generation, OCR, and math through questions that require one or more of these capabilities.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmvet:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmvet", "languages": [], "modality": "multimodal", "name": "MMVet", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.19, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.7619, "raw_min": 0.671, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmvet:qwen2.5-vl-72b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500"}, "unit": null}, "slug": "llm-stats-mmvet", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "spatial_reasoning", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MM-Vet evaluation using GPT-4 Turbo for scoring. This variant of MM-Vet examines large multimodal models on complicated multimodal tasks requiring integrated capabilities across six core vision-language abilities: recognition, knowledge, spatial awareness, language generation, OCR, and math.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmvetgpt4turbo:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmvetgpt4turbo", "languages": [], "modality": "multimodal", "name": "MMVetGPT4Turbo", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.74, "raw_min": 0.74, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmvetgpt4turbo:qwen2-vl-72b", "reported_date": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500"}, "unit": null}, "slug": "llm-stats-mmvetgpt4turbo", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMVU (Multimodal Multi-disciplinary Video Understanding) is a benchmark for evaluating multimodal models on video understanding tasks across multiple disciplines, testing comprehension and reasoning capabilities on video content.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mmvu:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mmvu", "languages": [], "modality": "multimodal", "name": "MMVU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.4, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.804, "raw_min": 0.723, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mmvu:kimi-k2.5", "reported_date": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500"}, "unit": null}, "slug": "llm-stats-mmvu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500", "unit": null}, {"aliases": ["MMVU"], "categories": ["vision"], "collected_at": null, "description": "Video understanding benchmark. For non-video-native models, uses 1 fps frame extraction. Scores depend on frame sampling strategy and context length.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mmvu\u0000zai_glm_5_3_flash_model_card\u0000mmvu\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mmvu\u0000zai_glm_5_3_flash_model_card\u0000mmvu\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:mmvu", "languages": [], "modality": null, "name": "MMVU", "openness": "unknown", "publisher": null, "released": "2025-06-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mmvu", "source_url": "https://github.com/MMVU/MMVU"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.5, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 80.5, "raw_min": 80.5, "source_reference": {"obs_id": "curated\u0000mmvu\u0000zai_glm_5_3_flash_model_card\u0000mmvu\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "observation_id": "curated\u0000mmvu\u0000zai_glm_5_3_flash_model_card\u0000mmvu\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "mmvu", "source": "model_reports", "source_url": "https://github.com/MMVU/MMVU", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "frontend_development", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MobileMiniWob++ SR (Success Rate) is an adaptation of the MiniWob++ web interaction benchmark for mobile Android environments within AndroidWorld. It comprises 92 web interaction tasks adapted for touch-based mobile interfaces, evaluating agents' ability to navigate and interact with web applications on mobile devices.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mobileminiwob++-sr:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mobileminiwob++-sr", "languages": [], "modality": "multimodal", "name": "MobileMiniWob++_SR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.914, "raw_min": 0.68, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mobileminiwob++-sr:qwen2.5-vl-7b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500"}, "unit": null}, "slug": "llm-stats-mobileminiwob-sr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MobileWorld is a benchmark for evaluating multimodal agents on real mobile-device tasks, testing GUI grounding, navigation, and multi-step task completion in mobile environments.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mobileworld:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mobileworld", "languages": [], "modality": "multimodal", "name": "MobileWorld", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.8, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.778, "raw_min": 0.7, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mobileworld:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500"}, "unit": null}, "slug": "llm-stats-mobileworld", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "科学智能", "AI for Science", "逻辑推理", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MolPuzzle is a benchmark for evaluating MLM's reasoning ability, comprising 234 instances of structure elucidation, which feature over 18,000 QA samples. MolPuzzle用于考察MLM的推理能力，包含234个结构解析实例以及超过18000个QA样本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1273", "languages": [], "modality": null, "name": "MolPuzzle", "openness": "open", "publisher": "University of Notre Dame", "released": "2024-09-26", "released_reference": {"basis": "paper_first_version", "note": "Public publication date on the original NeurIPS submission. The separately imported bibliographic record's year-boundary date is not used.", "source_key": "opencompass:1273", "source_url": "https://openreview.net/forum?id=t1mAXb4Cop"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1273-molpuzzle", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MolPuzzle", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "Stereo Conversion", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "For evaluating stereo image conversion, it provides test data for five different scenarios: animation, indoor, outdoor, complex, and simple, with a total of approximately 2,500 test sample pairs. It also offers evaluation metrics for assessing the stereo effect. 用于测评立体影像转换，提供了动画，室内，室外，复杂，简单共五种场景的测试数据，总共约2500对测试样本。并提供了用于测评立体效果的评价指标-立体交并比。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1704", "languages": [], "modality": null, "name": "Mono2Stereo", "openness": "restricted", "publisher": null, "released": "2025-03-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1704-mono2stereo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Mono2Stereo", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "逻辑推理", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MORSE-500 introduces 500 programmatically generated videos testing 6 reasoning types: abstract, physical, planning, spatial, temporal, mathematical. Its controllable generation (via Manim, Matplotlib, generative models etc.) enables scalable difficulty, designed to evolve as SOTA models improve. 当前多模态推理基准存在三大不足：依赖静态图像、偏重数学解题、易饱和。为此，我们提出MORSE-500视频基准：包含500个脚本化视频片段，涵盖抽象、物理、规划、空间、时间、数学六类推理问题。其核心在于程序化生成（使用Manim、Matplotlib等），可精确控制视觉复杂度、干扰物密度和时序动态，从而系统化提升难度。与易过时的静态基准不同，MORSE-500具备可持续演进能力，其可控生成流程能无限创建新挑战实例。在顶尖模型（Gemini 2.5 Pro、OpenAI o3等）上的测试揭示了显著性能差距，尤其在抽象和规划任务上。我们开源数据集、生成脚本及评估工具，以促进透明、可复现的前沿研究。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1999", "languages": [], "modality": "multimodal", "name": "MORSE-500", "openness": "unknown", "publisher": "University of Maryland", "released": "2025-06-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1999-morse-500", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MORSE-500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MotionBench is a benchmark for evaluating multimodal models on motion understanding in videos, testing the ability to comprehend temporal dynamics, movement patterns, and action sequences.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:motionbench:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:motionbench", "languages": [], "modality": "multimodal", "name": "MotionBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.9, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.749, "raw_min": 0.704, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:motionbench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500"}, "unit": null}, "slug": "llm-stats-motionbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MotionBench, is a comprehensive evaluation benchmark designed to assess the fine-grained motion comprehension of video understanding models. MotionBench是一个综合性的评估基准，旨在评估视频理解模型的细粒度运动理解能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1682", "languages": [], "modality": "multimodal", "name": "MotionBench", "openness": "restricted", "publisher": "THUDM", "released": "2025-01-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1682-motionbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MotionBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "物理智能", "Embodied AI", "跨模态推理", "Cross-modal Reasoning", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MotionMillion: The largest open-sourced 3D human motion dataset with text annotation, including 2.5k hours 1.9M episodes. MotionMillion：目前最大的开源带文本标注的 3D 人体动作数据集，包含 2500 小时、190 万个片段。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": true, "key": "opencompass:2154", "languages": [], "modality": "multimodal", "name": "MotionMillion", "openness": "restricted", "publisher": "InternRobotics", "released": "2025-07-26", "released_reference": {"basis": "dataset_published", "note": "The authors explicitly date the dataset release July 26; the July 3 code release and July 9 paper are separate events.", "source_key": "opencompass:2154", "source_url": "https://github.com/VankouF/MotionMillion-Codes"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2154-motionmillion", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MotionMillion", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Welcome to the dataset page for the Meta-Reasoning Benchmark associated with our recent publication `Mr-Ben: A Comprehensive Meta-Reasoning Benchmark for Large Language Models`. We have provided a demo evaluate script for you to try out benchmark in mere two steps. We encourage everyone to try out our benchmark in the SOTA models and return its results to us. We would be happy to include it in the eval_results and update the evaluation tables below for you. 本工作联合MIT,清华,剑桥等知名院校, 提出了一个评测大语言模型对复杂问题的推理过程的“阅卷”批改能力的评测数据集，有别于以前的以结果匹配为评测模式的数据集MR-Ben，我们的数据集基于GSM8K[1], MMLU[2], LogiQA[3], MHPP[4]等数据集经由细致的高水平人工标注构建而成，显著地增加了难度及区分度。我们细致地分析了包括claude3.5, GPT4-Turbo, Kimi, Zhipu, Yi-Large, Qwen2, DeepseekCoderv2 等国内外一线的大语言模型，发现开源的模型在复杂推理的场景下有望追上顶尖的闭源模型。该评测数据集的所有数据均已开源，并且支持一键评测。欢迎所有做大模型训练的小伙伴向我们分享你的评测结果，我们会及时更新榜单。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:930", "languages": [], "modality": null, "name": "MR-Ben-Meta-Reasoning-Benchmark", "openness": "open", "publisher": null, "released": "2024-06-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-930-mr-ben-meta-reasoning-benchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MR-Ben-Meta-Reasoning-Benchmark", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MR-GSM8K is a challenging benchmark designed to evaluate the meta-reasoning capabilities of state-of-the-art Large Language Models (LLMs). It goes beyond traditional evaluation metrics by focusing on the reasoning process rather than just the final answer. MR-GSM8K 是一个旨在评估最先进大型语言模型（LLMs）元推理能力的挑战性基准。它超越了传统的评估指标，专注于推理过程而非仅仅关注最终答案，从而对模型的认知能力进行更细致的评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1587", "languages": [], "modality": null, "name": "MR-GSM8K", "openness": "unknown", "publisher": "HKU, Tencent AI Lab, etc.", "released": "2023-12-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1587-mr-gsm8k", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MR-GSM8K", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "VQA", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MRAG-Bench consists of 16,130 images and 1,353 human-annotated multiple-choice questions across 9 distinct scenarios, providing a robust and systematic evaluation of Large Vision Language Model (LVLM)’s vision-centric multimodal retrieval-augmented generation (RAG) abilities. MRAG-Bench 包含 16,130 张图片和 1,353 个跨越 9 个不同场景的人标注多选题，为大型视觉语言模型（LVLM）的视觉中心多模态检索增强生成（RAG）能力提供了稳健和系统的评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1586", "languages": [], "modality": "multimodal", "name": "MRAG-Bench", "openness": "unknown", "publisher": "UCLA, Stanford University", "released": "2024-10-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1586-mrag-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MRAG-Bench", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR (Multi-Round Coreference Resolution) is a synthetic long-context reasoning task where models must navigate long conversations to reproduce specific model outputs. It tests the ability to distinguish between similar requests and reason about ordering while maintaining attention across extended contexts.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr:gemini-1.5-flash-8b", "reported_at": "2024-03-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr", "languages": [], "modality": "text", "name": "MRCR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.0, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.93, "raw_min": 0.32, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr:gemini-2.5-pro", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500", "unit": null}, {"aliases": ["MRCR", "OpenAI-MRCR", "MRCR v2"], "categories": ["long_context"], "collected_at": null, "description": "Needle position and generation seed change the result.", "evidence_summary": {"document_count": 8, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mrcr\u0000google_gemini_3_1_pro_model_card\u0000mrcr_v2\u0000Thinking (High), 128k (average)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "first_score_reported_at": "2026-02-19", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mrcr\u0000google_gemini_3_1_pro_model_card\u0000mrcr_v2\u0000Thinking (High), 128k (average)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:mrcr", "languages": [], "modality": null, "name": "MRCR", "openness": "unknown", "publisher": null, "released": "2025-04-14", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mrcr", "source_url": "https://huggingface.co/datasets/openai/mrcr"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.9, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 84.9, "raw_min": 26.3, "source_reference": {"obs_id": "curated\u0000mrcr\u0000google_gemini_3_1_pro_model_card\u0000mrcr_v2\u0000Thinking (High), 128k (average)\u0000Gemini 3.1 Pro", "observation_id": "curated\u0000mrcr\u0000google_gemini_3_1_pro_model_card\u0000mrcr_v2\u0000Thinking (High), 128k (average)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "reported_date": "2026-02-19", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "unit": "percent"}, "slug": "mrcr", "source": "model_reports", "source_url": "https://huggingface.co/datasets/openai/mrcr", "unit": "percent"}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR (Multi-Round Coreference Resolution) at 128K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 2 items to retrieve.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-128k-(2-needle):minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-128k-(2-needle)", "languages": [], "modality": "text", "name": "MRCR 128K (2-needle)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 28.62, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.2862, "raw_min": 0.2862, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-128k-(2-needle):minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-128k-2-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR (Multi-Round Coreference Resolution) at 128K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 4 items to retrieve.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-128k-(4-needle):minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-128k-(4-needle)", "languages": [], "modality": "text", "name": "MRCR 128K (4-needle)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 19.62, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.1962, "raw_min": 0.1962, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-128k-(4-needle):minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-128k-4-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR (Multi-Round Coreference Resolution) at 128K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 8 items to retrieve.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-128k-(8-needle):minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-128k-(8-needle)", "languages": [], "modality": "text", "name": "MRCR 128K (8-needle)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.904, "raw_min": 0.1012, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-128k-(8-needle):qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-128k-8-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR 1M is a variant of the Multi-Round Coreference Resolution benchmark designed for testing extremely long context capabilities with approximately 1 million tokens. It evaluates models' ability to maintain reasoning and attention across ultra-long conversations.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-1m:gemini-2.0-flash-lite", "reported_at": "2025-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-1m", "languages": [], "modality": "text", "name": "MRCR 1M", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.5, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.835, "raw_min": 0.58, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-1m:deepseek-v4-pro-max", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-1m", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR 1M (pointwise) is a variant of the Multi-Round Coreference Resolution benchmark that uses pointwise evaluation for ultra-long contexts (~1M tokens). This version evaluates each response independently rather than comparatively, testing models' absolute performance on long-context reasoning tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-1m-(pointwise):gemini-2.5-pro", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-1m-(pointwise)", "languages": [], "modality": "text", "name": "MRCR 1M (pointwise)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.89999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.829, "raw_min": 0.829, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-1m-(pointwise):gemini-2.5-pro", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-1m-pointwise", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR (Multi-Round Coreference Resolution) at 64K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 2 items to retrieve.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-64k-(2-needle):minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-64k-(2-needle)", "languages": [], "modality": "text", "name": "MRCR 64K (2-needle)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 29.770000000000003, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.2977, "raw_min": 0.2977, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-64k-(2-needle):minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-64k-2-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR (Multi-Round Coreference Resolution) at 64K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 4 items to retrieve.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-64k-(4-needle):minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-64k-(4-needle)", "languages": [], "modality": "text", "name": "MRCR 64K (4-needle)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 20.57, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.2057, "raw_min": 0.2057, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-64k-(4-needle):minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-64k-4-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR (Multi-Round Coreference Resolution) at 64K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 8 items to retrieve.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-64k-(8-needle):minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-64k-(8-needle)", "languages": [], "modality": "text", "name": "MRCR 64K (8-needle)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 16.56, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.1656, "raw_min": 0.1656, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-64k-(8-needle):minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-64k-8-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR v2 (Multi-Round Coreference Resolution version 2) is an enhanced version of the synthetic long-context reasoning task. It extends the original MRCR framework with improved evaluation criteria and additional complexity for testing models' ability to maintain attention and reasoning across extended contexts.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-v2:gemini-2.5-flash-lite", "reported_at": "2025-06-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-v2", "languages": [], "modality": "text", "name": "MRCR v2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.917, "raw_min": 0.166, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-v2:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR v2 (8-needle) is a variant of the Multi-Round Coreference Resolution benchmark that includes 8 needle items to retrieve from long contexts. This tests models' ability to simultaneously track and reason about multiple pieces of information across extended conversations.", "evidence_summary": {"document_count": 1, "model_count": 23, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-v2-(8-needle):gemma-3-27b-it", "reported_at": "2025-03-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-v2-(8-needle)", "languages": [], "modality": "text", "name": "MRCR v2 (8-needle)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 23, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.0, "display_multiplier": 100, "model_count": 23, "model_count_basis": "source_model_id", "numeric_count": 23, "raw_max": 0.97, "raw_min": 0.135, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-v2-(8-needle):gemini-3.7-flash", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-v2-8-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MRCR v2 8-needle variant evaluated on contexts from 512K to 1M tokens.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mrcr-v2-8-needle-512k-1m:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mrcr-v2-8-needle-512k-1m", "languages": [], "modality": "text", "name": "MRCR v2 (8-needle, 512K-1M)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.8, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.738, "raw_min": 0.413, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mrcr-v2-8-needle-512k-1m:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500"}, "unit": null}, "slug": "llm-stats-mrcr-v2-8-needle-512k-1m", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MS MARCO comprises of 1,010,916 anonymized questions—sampled from Bing’s search query logs—each with a human generated answer and 182,669 completely human rewritten generated answers. In addition, the dataset contains 8,841,823 passages. MS MARCO 数据集包含 1,010,916 个来自 Bing 的搜索查询日志的匿名问题，每个问题都有一个人工生成的答案和 182,669 个完全由人重写的生成答案。此外，该数据集还包含从 3,563,535 个由 Bing 检索的网页文档中提取的 8,841,823 个段落，这些段落提供了策划自然语言答案所需的信息。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1102", "languages": [], "modality": null, "name": "MS_MARCO", "openness": "unknown", "publisher": "Microsoft AI & Research", "released": "2018-10-31", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1102-ms-marco", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MS_MARCO", "unit": null}, {"aliases": [], "categories": ["reasoning", "knowledge", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MSQA is a multilingual question-answering benchmark that measures knowledge and reasoning across a diverse set of languages.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:msqa:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:msqa", "languages": [], "modality": "text", "name": "MSQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 50.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.502, "raw_min": 0.42, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:msqa:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500"}, "unit": null}, "slug": "llm-stats-msqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MT-AIME 2025 is Cohere's internal multilingual translation of AIME 2025, evaluated for Arabic, Japanese, and Korean in the Command A+ release.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mt-aime-2025:command-a-plus-05-2026", "reported_at": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mt-aime-2025", "languages": [], "modality": "text", "name": "MT-AIME 2025", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.86, "raw_min": 0.86, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mt-aime-2025:command-a-plus-05-2026", "reported_date": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500"}, "unit": null}, "slug": "llm-stats-mt-aime-2025", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "roleplay", "general", "communication", "creativity"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MT-Bench is a challenging multi-turn benchmark that measures the ability of large language models to engage in coherent, informative, and engaging conversations. It uses strong LLMs as judges for scalable and explainable evaluation of multi-turn dialogue capabilities.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mt-bench:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mt-bench", "languages": [], "modality": "text", "name": "MT-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 8.99, "display_multiplier": 1, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 8.99, "raw_min": 0.0899, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mt-bench:hermes-3-70b", "reported_date": "2024-08-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-mt-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "ACL 2024", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MT-Bench-101 is specifically designed to evaluate the finegrained abilities of LLMs in multi-turn dialogues. MT-Bench-101 专门设计用于评估 LLMs 在多轮对话中的细粒度能力。通过对真实多轮对话数据的详细分析，构建了一个三层级的能力分类法，涵盖 1388 个多轮对话中的 4208 个轮次，涉及 13 种不同的任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1172", "languages": [], "modality": null, "name": "MT-Bench-101", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-06-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1172-mt-bench-101", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MT-Bench-101", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "知识", "Knowledge", "理解", "Understanding", "Chinese Medicine", "Benchmark", "TCM", "科学智能", "AI for Science", "逻辑推理", "知识储备", "语言理解", "Comprehension", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "We introduce MTCMB-a Multi-Task Benchmark for Evaluating LLMs on TCM Knowledge, Reasoning, and Safety. Developed in collaboration with certified TCM experts, MTCMB comprises 12 sub-datasets spanning five major categories: knowledge QA, language understanding, diagnostic reasoning, formula generation 大语言模型在中医领域的应用日益增多，到底大语言模型在这一古老的学科表现如何？虽然有一些零碎的评测数据集，但大都以单选题、病案分析、粗糙的医患对话为主，很难全面评估大语言模型在中医领域的实际能力。近日，中山大学联合湖南中医药大学等团队推出全球首个中医多任务评测基准（Benchmark） —— MTCMB: A Multi-Task Benchmark Framework for Evaluating LLMs on Knowledge, Reasoning, and Safety in Traditional Chinese Medicine", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1929", "languages": ["Chinese"], "modality": null, "name": "MTCMB", "openness": "restricted", "publisher": "Sun Yat-sen University, Hunan University of Chinese Medicine", "released": "2025-05-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1929-mtcmb", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MTCMB", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "文本嵌入", "大语言模型", "LLM", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MTEB is a benchmark for evaluating text embedding models, aiming to enhance long-term usability and reproducibility, focusing on task extensibility, data integrity validation, and result generalizability. MTEB 是一个评估文本嵌入模型的基准，旨在提升其长期可用性和可复现性，涵盖任务扩展性、数据完整性验证和结果通用性等维度。该基准包含包括分类、检索和聚类在内的多种任务，覆盖多个语言和领域。MTEB 引入了自动化测试管道、持续集成框架和社区贡献机制，以支持基准的可扩展性和质量控制。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2034", "languages": ["Chinese", "German", "Multilingual"], "modality": null, "name": "MTEB", "openness": "unknown", "publisher": "Zendesk , Esker , INSA Lyon , LIRIS , Aarhus University , ITMO University", "released": "2025-06-26", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2034-mteb", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MTEB", "unit": null}, {"aliases": ["MTOB", "Machine Translation from One Book"], "categories": ["long_context"], "collected_at": null, "description": "Translation from a single grammar book, reported as half-book and full-book variants that are not interchangeable.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:mtob", "languages": [], "modality": null, "name": "MTOB", "openness": "unknown", "publisher": null, "released": "2023-09-28", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mtob", "source_url": "https://arxiv.org/abs/2309.16575"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "mtob", "source": "model_reports", "source_url": "https://arxiv.org/abs/2309.16575", "unit": null}, {"aliases": [], "categories": ["multimodal", "text-to-image", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MTVQA (Multilingual Text-Centric Visual Question Answering) is the first benchmark featuring high-quality human expert annotations across 9 diverse languages, consisting of 6,778 question-answer pairs across 2,116 images. It addresses visual-textual misalignment problems in multilingual text-centric VQA.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mtvqa:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mtvqa", "languages": [], "modality": "multimodal", "name": "MTVQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 30.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.309, "raw_min": 0.309, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mtvqa:qwen2-vl-72b", "reported_date": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500"}, "unit": null}, "slug": "llm-stats-mtvqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "VQA", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MTVQA is designed to evaluate LMMs' multilingual text understanding, featuring high-quality human expert annotations across 9 diverse languages. MTVQA用于评估多模态大模型理解多语言文本的能力，包含来自9种语言的由人类专家注释的高质量数据。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1502", "languages": ["English", "Chinese", "Multilingual"], "modality": "multimodal", "name": "MTVQA", "openness": "restricted", "publisher": "ByteDance", "released": "2024-06-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1502-mtvqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MTVQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive benchmark for robust multi-image understanding capabilities of multimodal LLMs. Consists of 12 diverse multi-image tasks involving 10 categories of multi-image relations (e.g., multiview, temporal relations, narrative, complementary). Comprises 11,264 images and 2,600 multiple-choice questions created in a pairwise manner, where each standard instance is paired with an unanswerable variant for reliable assessment.", "evidence_summary": {"document_count": 1, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:muirbench:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:muirbench", "languages": [], "modality": "multimodal", "name": "MuirBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 12, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.30000000000001, "display_multiplier": 100, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 12, "raw_max": 0.803, "raw_min": 0.583, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:muirbench:qwen3-vl-32b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500"}, "unit": null}, "slug": "llm-stats-muirbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "communication"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MultiChallenge is a realistic multi-turn conversation evaluation benchmark that challenges frontier LLMs across four key categories: instruction retention (maintaining instructions throughout conversations), inference memory (recalling and connecting details from previous turns), reliable versioned editing (adapting to evolving instructions during collaborative editing), and self-coherence (avoiding contradictions in responses). The benchmark evaluates models on sustained, contextually complex dialogues across diverse topics including travel planning, technical documentation, and professional communication.", "evidence_summary": {"document_count": 1, "model_count": 29, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multichallenge:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:multichallenge", "languages": [], "modality": "text", "name": "Multi-Challenge", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 29, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.7, "display_multiplier": 100, "model_count": 29, "model_count_basis": "source_model_id", "numeric_count": 29, "raw_max": 0.777, "raw_min": 0.15, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multichallenge:nova-2-pro", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500"}, "unit": null}, "slug": "llm-stats-multichallenge", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "structured_output", "instruction_following", "language", "communication"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Multi-IF benchmarks LLMs on multi-turn and multilingual instruction following. It expands upon IFEval by incorporating multi-turn sequences and translating English prompts into 7 other languages, resulting in 4,501 multilingual conversations with three turns each. The benchmark reveals that current leading LLMs struggle with maintaining accuracy in multi-turn instructions and shows higher error rates for non-Latin script languages.", "evidence_summary": {"document_count": 1, "model_count": 23, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multi-if:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:multi-if", "languages": [], "modality": "text", "name": "Multi-IF", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 23, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.60000000000001, "display_multiplier": 100, "model_count": 23, "model_count_basis": "source_model_id", "numeric_count": 23, "raw_max": 0.806, "raw_min": 0.373, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multi-if:qwen3-235b-a22b-thinking-2507", "reported_date": "2025-07-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500"}, "unit": null}, "slug": "llm-stats-multi-if", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multilingual benchmark for issue resolving that evaluates Large Language Models' ability to resolve software issues across diverse programming ecosystems. Covers 7 programming languages (Java, TypeScript, JavaScript, Go, Rust, C, and C++) with 1,632 high-quality instances carefully annotated by 68 expert annotators. Addresses limitations of existing benchmarks that focus almost exclusively on Python.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multi-swe-bench:qwen3-coder-480b-a35b-instruct", "reported_at": "2025-01-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:multi-swe-bench", "languages": [], "modality": "text", "name": "Multi-SWE-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 52.7, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.527, "raw_min": 0.258, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multi-swe-bench:minimax-m2.7", "reported_date": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-multi-swe-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500", "unit": null}, {"aliases": ["MultiChallenge"], "categories": ["instruction_following"], "collected_at": null, "description": "Multi-turn instruction retention, graded by an LLM judge.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000multichallenge\u0000qwen3_5_model_card\u0000multichallenge\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000multichallenge\u0000qwen3_5_model_card\u0000multichallenge\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:multichallenge", "languages": [], "modality": null, "name": "MultiChallenge", "openness": "unknown", "publisher": null, "released": "2025-01-29", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:multichallenge", "source_url": "https://arxiv.org/abs/2501.17399"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.6, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 67.6, "raw_min": 67.6, "source_reference": {"obs_id": "curated\u0000multichallenge\u0000qwen3_5_model_card\u0000multichallenge\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000multichallenge\u0000qwen3_5_model_card\u0000multichallenge\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "multichallenge", "source": "model_reports", "source_url": "https://arxiv.org/abs/2501.17399", "unit": "percent"}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MultiLF benchmark", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multilf:qwen3-235b-a22b", "reported_at": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:multilf", "languages": [], "modality": "text", "name": "MultiLF", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.73, "raw_min": 0.719, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multilf:qwen3-32b", "reported_date": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500"}, "unit": null}, "slug": "llm-stats-multilf", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Multilingual Grade School Math (MGSM) benchmark evaluates language models' chain-of-thought reasoning abilities across ten typologically diverse languages. Contains 250 grade-school math problems manually translated from GSM8K dataset into languages including Bengali and Swahili.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multilingual-mgsm-(cot):llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:multilingual-mgsm-(cot)", "languages": [], "modality": "text", "name": "Multilingual MGSM (CoT)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.60000000000001, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.916, "raw_min": 0.689, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multilingual-mgsm-(cot):llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500"}, "unit": null}, "slug": "llm-stats-multilingual-mgsm-cot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMLU-ProX is a comprehensive multilingual benchmark covering 29 typologically diverse languages, building upon MMLU-Pro. Each language version consists of 11,829 identical questions enabling direct cross-linguistic comparisons. The benchmark evaluates large language models' reasoning capabilities across linguistic and cultural boundaries through challenging, reasoning-focused questions with 10 answer choices.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multilingual-mmlu:o3-mini", "reported_at": "2025-01-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:multilingual-mmlu", "languages": [], "modality": "text", "name": "Multilingual MMLU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.7, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.807, "raw_min": 0.493, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multilingual-mmlu:o3-mini", "reported_date": "2025-01-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500"}, "unit": null}, "slug": "llm-stats-multilingual-mmlu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MultiLoKo is a multilingual knowledge benchmark, covering 30 languages plus English. MultiLoKo是一个多语言知识基准，涵盖30种语言及英语。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1751", "languages": ["English", "Multilingual"], "modality": null, "name": "MultiLoKo", "openness": "unknown", "publisher": "Meta", "released": "2025-04-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1751-multiloko", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MultiLoKo", "unit": null}, {"aliases": [], "categories": ["language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MultiPL-E is a scalable and extensible system for translating unit test-driven code generation benchmarks to multiple programming languages. It extends HumanEval and MBPP Python benchmarks to 18 additional programming languages, enabling evaluation of neural code generation models across diverse programming paradigms and language features.", "evidence_summary": {"document_count": 1, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multipl-e:qwen2-72b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:multipl-e", "languages": [], "modality": "text", "name": "MultiPL-E", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 13, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.9, "display_multiplier": 100, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 13, "raw_max": 0.879, "raw_min": 0.591, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multipl-e:qwen3-235b-a22b-instruct-2507", "reported_date": "2025-07-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500"}, "unit": null}, "slug": "llm-stats-multipl-e", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500", "unit": null}, {"aliases": [], "categories": ["language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MultiPL-E is a scalable and extensible approach to benchmarking neural code generation that translates unit test-driven code generation benchmarks across multiple programming languages. It extends the HumanEval benchmark to 18 additional programming languages, enabling evaluation of code generation models across diverse programming paradigms and providing insights into how models generalize programming knowledge across language boundaries.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multipl-e-humaneval:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:multipl-e-humaneval", "languages": [], "modality": "text", "name": "Multipl-E HumanEval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.752, "raw_min": 0.508, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multipl-e-humaneval:llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500"}, "unit": null}, "slug": "llm-stats-multipl-e-humaneval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MultiPL-E extends the Mostly Basic Python Problems (MBPP) benchmark to 18+ programming languages for evaluating multilingual code generation capabilities. MBPP contains 974 crowd-sourced programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality. Each problem includes a task description, code solution, and automated test cases.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:multipl-e-mbpp:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:multipl-e-mbpp", "languages": [], "modality": "text", "name": "Multipl-E MBPP", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.657, "raw_min": 0.524, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:multipl-e-mbpp:llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500"}, "unit": null}, "slug": "llm-stats-multipl-e-mbpp", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MusicCaps is a dataset composed of 5,521 music examples, each labeled with an English aspect list and a free text caption written by musicians. The dataset contains 10-second music clips from AudioSet paired with rich textual descriptions that capture sonic qualities and musical elements like genre, mood, tempo, instrumentation, and rhythm. Created to support research in music-text understanding and generation tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:musiccaps:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:musiccaps", "languages": [], "modality": "multimodal", "name": "MusicCaps", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 32.800000000000004, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.328, "raw_min": 0.328, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:musiccaps:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500"}, "unit": null}, "slug": "llm-stats-musiccaps", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MuSR (Multistep Soft Reasoning) is a benchmark for evaluating language models on multistep soft reasoning tasks specified in natural language narratives. Created through a neurosymbolic synthetic-to-natural generation algorithm, it generates complex reasoning scenarios like murder mysteries roughly 1000 words in length that challenge current LLMs including GPT-4. The benchmark tests chain-of-thought reasoning capabilities across domains involving commonsense reasoning about physical and social situations.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:musr:hermes-3-70b", "reported_at": "2024-08-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:musr", "languages": [], "modality": "text", "name": "MuSR", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.764, "raw_min": 0.5067, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:musr:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500"}, "unit": null}, "slug": "llm-stats-musr", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "spatial_reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive multi-modal video understanding benchmark covering 20 challenging video tasks that require temporal understanding beyond single-frame analysis. Tasks span from perception to cognition, including action recognition, temporal reasoning, spatial reasoning, object interaction, scene transition, and counterfactual inference. Uses a novel static-to-dynamic method to systematically generate video tasks from existing annotations.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:mvbench:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:mvbench", "languages": [], "modality": "multimodal", "name": "MVBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.6, "display_multiplier": 100, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 0.766, "raw_min": 0.687, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:mvbench:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500"}, "unit": null}, "slug": "llm-stats-mvbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500", "unit": null}, {"aliases": ["MVbench", "MVBench"], "categories": ["vision"], "collected_at": null, "description": "Video understanding benchmark. For non-video-native models, uses 1 fps frame extraction. Scores depend on frame sampling strategy and context length.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000mvbench\u0000zai_glm_5_3_flash_model_card\u0000mvbench\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000mvbench\u0000zai_glm_5_3_flash_model_card\u0000mvbench\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:mvbench", "languages": [], "modality": null, "name": "MVbench", "openness": "unknown", "publisher": null, "released": "2025-06-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:mvbench", "source_url": "https://github.com/MVBench/MVBench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.8, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 77.8, "raw_min": 77.8, "source_reference": {"obs_id": "curated\u0000mvbench\u0000zai_glm_5_3_flash_model_card\u0000mvbench\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "observation_id": "curated\u0000mvbench\u0000zai_glm_5_3_flash_model_card\u0000mvbench\u0000temp 1.0, top_p 0.95, max context 256K, 1 fps frame-extraction for non-video-native models\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "mvbench", "source": "model_reports", "source_url": "https://github.com/MVBench/MVBench", "unit": "percent"}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "视频理解", "Video Understanding", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MVBench can test MLLMs' temporal understanding in the dynamic video tasks. It covers 20 challenging video tasks that cannot be effectively solved with a single frame. MVBench用于评估多模态大模型在动态视频任务中的时间理解能力，由20个单帧内容无法有效解决的挑战性的视频任务组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1509", "languages": [], "modality": "multimodal", "name": "MVBench", "openness": "open", "publisher": "Chinese Academy of Sciences", "released": "2023-11-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1509-mvbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MVBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "语言", "Language", "多模态模型", "VLM", "图像理解", "Image Understanding", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MVL-SIB is a multilingual dataset that provides image-sentence pairs spanning 205 languages and 7 topical categories (entertainment, geography, health, politics, science, sports, travel). It was constructed by extending the SIB-200 benchmark. MVL-SIB 是一个多语言数据集，提供了涵盖 205 种语言和 7 个主题类别的图像-句子对（ entertainment ， geography ， health ， politics ， science ， sports ， travel ）。它通过扩展 SIB-200 基准构建而成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1537", "languages": ["English", "Multilingual"], "modality": "multimodal", "name": "MVL-SIB", "openness": "unknown", "publisher": "University of Würzburg, University of Hamburg", "released": "2025-02-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1537-mvl-sib", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MVL-SIB", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "理解", "Understanding", "多模态模型", "VLM", "逻辑推理", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "MVPBench, a curated benchmark designed to rigorously evaluate visual physical reasoning through the lens of visual CoT. Each example features interleaved multi-image inputs and demands not only the correct final answer but also a coherent, step-by-step reasoning path grounded in evolving visual cues MVPBench专注于视觉物理推理中的视觉链式思维（CoT）能力评估。该基准涵盖真实图像、多步逻辑与多条可行思维路径，每个样例均配有图像证据，要求模型在剥离文本提示依赖的前提下，不仅得出正确答案，还需正确完成每一个中间推理步骤，模拟人类的逐步图像推理过程。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1905", "languages": [], "modality": "multimodal", "name": "MVPBench", "openness": "open", "publisher": "Central South University", "released": "2025-06-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1905-mvpbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MVPBench", "unit": null}, {"aliases": [], "categories": ["agents", "code", "systems"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NanoGPT is an OpenAI AI-self-improvement evaluation that measures whether models can optimize training recipes for small GPT-style models, part of the suite tracking progress toward accelerating internal research.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nanogpt:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nanogpt", "languages": [], "modality": "text", "name": "NanoGPT", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 14.499999999999998, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.145, "raw_min": 0.0166, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nanogpt:gpt-5.6-terra", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500"}, "unit": null}, "slug": "llm-stats-nanogpt", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Natural Questions is a question answering dataset featuring real anonymized queries issued to Google search engine. It contains 307,373 training examples where annotators provide long answers (passages) and short answers (entities) from Wikipedia pages, or mark them as unanswerable.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:natural-questions:gemma-2-27b-it", "reported_at": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:natural-questions", "languages": [], "modality": "text", "name": "Natural Questions", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 34.5, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.345, "raw_min": 0.155, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:natural-questions:gemma-2-27b-it", "reported_date": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500"}, "unit": null}, "slug": "llm-stats-natural-questions", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NaturalCodeBench (NCB) is a challenging code benchmark designed to mirror the complexity and variety of real-world coding tasks. It comprises 402 high-quality problems in Python and Java, selected from natural user queries from online coding services, covering 6 different domains.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:natural2code:gemini-1.5-flash-8b", "reported_at": "2024-03-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:natural2code", "languages": [], "modality": "text", "name": "Natural2Code", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.9, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.929, "raw_min": 0.56, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:natural2code:gemini-2.0-flash", "reported_date": "2024-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500"}, "unit": null}, "slug": "llm-stats-natural2code", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "ACL 2024", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "NaturalCodeBench is challenging code benchmark designed to mirror the complexity and variety of scenarios in real coding tasks. NCB comprises 402 high-quality problems in Python and Java, meticulously selected from natural user queries from online coding services, covering 6 different domains. NaturalCodeBench 是一个具有挑战性的代码基准，旨在反映真实编码任务中的复杂性和多样性。NaturalCodeBench 包含 402 个高质量的 Python 和 Java 问题，这些问题是从在线编码服务的自然用户查询中精心挑选的，涵盖了 6 个不同的领域。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1081", "languages": [], "modality": null, "name": "NaturalCodeBench", "openness": "restricted", "publisher": "THUDM", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1081-naturalcodebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NaturalCodeBench", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "科学智能", "AI for Science", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "NATURALPROOFS is a multi-domain corpus of mathematical statements and their proofs, written in natural mathematical language. NATURALPROOFS unifies broad coverage, deep coverage, and low-resource mathematical sources, allowing for evaluating both in-distribution and zero-shot generalization. NATURALPROOFS 是一个多领域的数学语句及其证明的语料库，采用自然数学语言编写。整合了广泛覆盖、深入覆盖和低资源数学来源，便于评估分布内和零样本泛化的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1117", "languages": [], "modality": null, "name": "NaturalProofs", "openness": "unknown", "publisher": "Allen Institute for Artificial Intelligence", "released": "2021-06-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1117-naturalproofs", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NaturalProofs", "unit": null}, {"aliases": [], "categories": ["创作", "Creation", "ACL 2024", "大语言模型", "LLM", "语言生成", "Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "NewsBench is a novel evaluation framework to systematically assess the capabilities of Large Language Models (LLMs) for editorial capabilities in Chinese journalism. NewsBench 是一个新颖的评估框架，旨在系统性地评估大型语言模型在中文新闻编辑能力上的表现。构建的基准数据集聚焦于写作能力的四个方面和安全遵循的六个方面，包含 1,267 个手动精心设计的测试样本，类型包括选择题和简答题，涵盖 24 个新闻领域的五项编辑任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1074", "languages": ["Chinese"], "modality": null, "name": "NewsBench", "openness": "unknown", "publisher": "IAAR-Shanghai", "released": "2024-06-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1074-newsbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NewsBench", "unit": null}, {"aliases": [], "categories": ["general", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NexusRaven benchmark for evaluating function calling capabilities of large language models in zero-shot scenarios across cybersecurity tools and API interactions", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nexus:llama-3.1-405b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nexus", "languages": [], "modality": "text", "name": "Nexus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 58.699999999999996, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.587, "raw_min": 0.343, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nexus:llama-3.1-405b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500"}, "unit": null}, "slug": "llm-stats-nexus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Multi-needle in a haystack benchmark for evaluating long-context comprehension capabilities of language models by testing retrieval of multiple target pieces of information from extended documents", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nih-multi-needle:llama-3.2-3b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nih-multi-needle", "languages": [], "modality": "text", "name": "NIH/Multi-needle", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.847, "raw_min": 0.847, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nih-multi-needle:llama-3.2-3b-instruct", "reported_date": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500"}, "unit": null}, "slug": "llm-stats-nih-multi-needle", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NL2Repo evaluates long-horizon coding capabilities including repository-level understanding, where models must generate or modify code across entire repositories from natural language specifications.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nl2repo:minimax-m2.7", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nl2repo", "languages": [], "modality": "text", "name": "NL2Repo", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.5, "display_multiplier": 100, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 0.615, "raw_min": 0.294, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nl2repo:deepseek-v4-pro-0813", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500"}, "unit": null}, "slug": "llm-stats-nl2repo", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500", "unit": null}, {"aliases": ["NL2Repo", "NL-to-Repo"], "categories": ["coding_agent"], "collected_at": null, "description": "Evaluates generating a repository from a natural language description. Temperature=1.0, top_p=1.0, max_new_tokens=64k under 1M context.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000nl2repo\u0000zai_glm_5_3_flash_model_card\u0000nl2repo\u0000temp 1.0, top_p 1.0, max_new_tokens 64k, 1M context, rule-based + LLM judge for hacking prevention\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000nl2repo\u0000zai_glm_5_3_flash_model_card\u0000nl2repo\u0000temp 1.0, top_p 1.0, max_new_tokens 64k, 1M context, rule-based + LLM judge for hacking prevention\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:nl2repo", "languages": [], "modality": null, "name": "NL2Repo", "openness": "unknown", "publisher": null, "released": "2025-06-15", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:nl2repo", "source_url": "https://github.com/nl2repo/nl2repo"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 58.9, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 58.9, "raw_min": 56.3, "source_reference": {"obs_id": "curated\u0000nl2repo\u0000tencent_hy4_preview\u0000nl2repo\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000nl2repo\u0000tencent_hy4_preview\u0000nl2repo\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "nl2repo", "source": "model_reports", "source_url": "https://github.com/nl2repo/nl2repo", "unit": "percent"}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NMOS evaluation benchmark for assessing model performance on specialized tasks", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nmos:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nmos", "languages": [], "modality": "text", "name": "NMOS", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.0451, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.0451, "raw_min": 0.0451, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nmos:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500"}, "unit": null}, "slug": "llm-stats-nmos", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:community:2256e9c9-b256-4444-b639-7cc3b1855d96", "languages": [], "modality": null, "name": "nolima", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-community-2256e9c9-b256-4444-b639-7cc3b1855d96", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A2256e9c9-b256-4444-b639-7cc3b1855d96?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NoLiMa evaluated at a 131072-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nolima-128k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nolima-128k", "languages": [], "modality": "text", "name": "NoLiMa 128K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 23.86, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.2386, "raw_min": 0.2386, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nolima-128k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500"}, "unit": null}, "slug": "llm-stats-nolima-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NoLiMa evaluated at a 32768-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nolima-32k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nolima-32k", "languages": [], "modality": "text", "name": "NoLiMa 32K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 54.54, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.5454, "raw_min": 0.5454, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nolima-32k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500"}, "unit": null}, "slug": "llm-stats-nolima-32k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NoLiMa evaluated at a 65536-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nolima-64k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nolima-64k", "languages": [], "modality": "text", "name": "NoLiMa 64K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 42.95, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.4295, "raw_min": 0.4295, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nolima-64k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500"}, "unit": null}, "slug": "llm-stats-nolima-64k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500", "unit": null}, {"aliases": [], "categories": ["general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "NOVA-63 is a multilingual evaluation benchmark covering 63 languages, designed to assess LLM performance across diverse linguistic contexts and tasks.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nova-63:qwen3.5-397b-a17b", "reported_at": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nova-63", "languages": [], "modality": "text", "name": "NOVA-63", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.099999999999994, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.591, "raw_min": 0.424, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nova-63:qwen3.5-397b-a17b", "reported_date": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500"}, "unit": null}, "slug": "llm-stats-nova-63", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Nondeterministic Polynomial-time Problem Challenge (NPPC), an ever-scaling reasoning benchmark for LLMs. 非确定性多项式时间问题挑战 （NPPC），这是一个不断扩展的 LLM 推理基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1783", "languages": [], "modality": null, "name": "NPPC", "openness": "unknown", "publisher": "The Hong Kong Polytechnic University, Carnegie Mellon University，etc.", "released": "2025-04-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1783-nppc", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NPPC", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Natural Questions (NQ) benchmark containing real user questions issued to Google search with answers found from Wikipedia, designed for training and evaluation of automatic question answering systems", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nq:granite-3.3-8b-base", "reported_at": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nq", "languages": [], "modality": "text", "name": "NQ", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 36.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.365, "raw_min": 0.365, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nq:granite-3.3-8b-base", "reported_date": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500"}, "unit": null}, "slug": "llm-stats-nq", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "knowledge", "大语言模型", "LLM", "知识储备", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "NQ (NaturalQuestion) corpus contains questions from real users, and it requires QA systems to read and comprehend an entire Wikipedia article that may or may not contain the answer to the question. The inclusion of real user questions, and the requirement that solutions should read an entire page to find the answer, cause NQ to be a more realistic and challenging task than prior QA datasets. NQ 数据集来自于真实用户的问题，它要求 QA 系统阅读和理解整个维基百科文章，这些文章可能包含也可能不包含问题的答案。由真实用户问题构成，以及需要阅读整个页面才能找到答案的要求，比以往的 QA 数据集更现实和更具挑战性的任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:513", "languages": [], "modality": null, "name": "NQ", "openness": "unknown", "publisher": null, "released": "2019-06-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-513-nq", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NQ", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "spatial", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multimodal benchmark for scene understanding and reasoning over the nuScenes autonomous driving domain.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:nuscene:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:nuscene", "languages": [], "modality": "multimodal", "name": "Nuscene", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 15.4, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.154, "raw_min": 0.146, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:nuscene:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500"}, "unit": null}, "slug": "llm-stats-nuscene", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "CoSyn-400K dataset contains 9 categories of synthetc text-rich images with 2.7M instruction-tuning data 通过代码引导的合成多模态数据生成扩展文本丰富图像理解，CoSyn-400K 数据集包含 9 类合成文本丰富图像，以及 270 万条指令微调数据。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1533", "languages": [], "modality": "multimodal", "name": "NutritionQA", "openness": "unknown", "publisher": "University of Pennsylvania,   Allen Institute for Artificial Intelligence", "released": "2025-02-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1533-nutritionqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NutritionQA", "unit": null}, {"aliases": [], "categories": ["3d", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Objectron evaluates 3D object detection and pose estimation capabilities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:objectron:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:objectron", "languages": [], "modality": "image", "name": "Objectron", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.712, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.712, "raw_min": 0.712, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:objectron:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500"}, "unit": null}, "slug": "llm-stats-objectron", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OCNLI is a Chinese natural language inference task, which requires to determine the logical relation between two sentences, with three relations: entailment, contradiction and neutral. OCNLI是一个中文自然语言推理任务，要求根据两个句子判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:525", "languages": ["Chinese"], "modality": null, "name": "OCNLI", "openness": "unknown", "publisher": null, "released": "2020-10-12", "released_reference": {"basis": "paper_first_version", "note": "First version introducing Original Chinese Natural Language Inference.", "source_key": "opencompass:525", "source_url": "https://arxiv.org/abs/2010.05444"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-525-ocnli", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OCNLI", "unit": null}, {"aliases": [], "categories": ["image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OCRBench: Comprehensive evaluation benchmark for assessing Optical Character Recognition (OCR) capabilities in Large Multimodal Models across text recognition, scene text VQA, and document understanding tasks", "evidence_summary": {"document_count": 1, "model_count": 24, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ocrbench:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:ocrbench", "languages": [], "modality": "multimodal", "name": "OCRBench", "openness": "unknown", "publisher": null, "released": "2024-01-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 24, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.30000000000001, "display_multiplier": 100, "model_count": 24, "model_count_basis": "source_model_id", "numeric_count": 24, "raw_max": 0.923, "raw_min": 0.792, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ocrbench:kimi-k2.5", "reported_date": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500"}, "unit": null}, "slug": "llm-stats-ocrbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "理解", "Understanding", "多模态模型", "VLM", "语言理解", "Comprehension", "图像理解", "Image Understanding", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OCRBench provides a comprehensive evaluation of Large Multimodal Models, such as GPT4V and Gemini, in various text-related visual tasks including Text Recognition, Scene Text-Centric Visual Question Answering (VQA), Document-Oriented VQA, Key Information Extraction (KIE), and Handwritten Mathematical Expression Recognition (HMER). OCRBench对 GPT4V 和 Gemini 等大型多模态模型在各种文本相关的视觉任务中的表现进行了全面的评估，包括文本识别、场景文本为中心的视觉问答 (VQA)、面向文档的 VQA、关键信息提取 (KIE) 和手写数学表达式识别 (HMER)。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:557", "languages": ["Multilingual"], "modality": null, "name": "OCRBench", "openness": "unknown", "publisher": null, "released": "2024-01-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-557-ocrbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OCRBench", "unit": null}, {"aliases": [], "categories": ["image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OCRBench v2 English subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with English text content", "evidence_summary": {"document_count": 1, "model_count": 14, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ocrbench-v2-(en):qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ocrbench-v2-(en)", "languages": [], "modality": "multimodal", "name": "OCRBench-V2 (en)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 14, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.4, "display_multiplier": 100, "model_count": 14, "model_count_basis": "source_model_id", "numeric_count": 14, "raw_max": 0.684, "raw_min": 0.367, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ocrbench-v2-(en):qwen3-vl-32b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500"}, "unit": null}, "slug": "llm-stats-ocrbench-v2-en", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OCRBench v2 Chinese subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with Chinese text content", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ocrbench-v2-(zh):qwen2.5-vl-32b", "reported_at": "2025-02-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ocrbench-v2-(zh)", "languages": [], "modality": "multimodal", "name": "OCRBench-V2 (zh)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.5, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.635, "raw_min": 0.558, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ocrbench-v2-(zh):qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500"}, "unit": null}, "slug": "llm-stats-ocrbench-v2-zh", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OCRBench v2: Enhanced large-scale bilingual benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with 10,000 human-verified question-answering pairs across 8 core OCR capabilities", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ocrbench-v2:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ocrbench-v2", "languages": [], "modality": "multimodal", "name": "OCRBench_V2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.10000000000001, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.671, "raw_min": 0.561, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ocrbench-v2:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-ocrbench-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Octopus coding benchmark for evaluating multi-language programming capabilities", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:octocodingbench:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:octocodingbench", "languages": [], "modality": "text", "name": "OctoCodingBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 26.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.261, "raw_min": 0.261, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:octocodingbench:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500"}, "unit": null}, "slug": "llm-stats-octocodingbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Object Detection in the Wild (ODinW) benchmark for evaluating object detection models' task-level transfer ability across diverse real-world datasets in terms of prediction accuracy and adaptation efficiency", "evidence_summary": {"document_count": 1, "model_count": 16, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:odinw:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:odinw", "languages": [], "modality": "image", "name": "ODinW", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 16, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.800000000000004, "display_multiplier": 100, "model_count": 16, "model_count_basis": "source_model_id", "numeric_count": 16, "raw_max": 0.518, "raw_min": 0.394, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:odinw:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500"}, "unit": null}, "slug": "llm-stats-odinw", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OfficeQA Pro evaluates AI models on professional knowledge-work questions and tasks drawn from real office workflows, including document analysis, spreadsheet reasoning, and information synthesis across business domains.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:officeqa-pro:gpt-5.5", "reported_at": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:officeqa-pro", "languages": [], "modality": "text", "name": "OfficeQA Pro", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.2, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.722, "raw_min": 0.451, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:officeqa-pro:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-officeqa-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500", "unit": null}, {"aliases": ["OfficeQA Pro", "OfficeQA"], "categories": ["vision"], "collected_at": null, "description": "Visual QA benchmark on professional document corpus (PDF without embedded text). Scores depend on context length and image resolution.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000office_qa_pro\u0000zai_glm_5_3_flash_model_card\u0000office_qa_pro\u0000Treasury Bulletin PDF corpus without embedded text access, temp 1.0, top_p 0.95, max context 512K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000office_qa_pro\u0000zai_glm_5_3_flash_model_card\u0000office_qa_pro\u0000Treasury Bulletin PDF corpus without embedded text access, temp 1.0, top_p 0.95, max context 512K\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:office_qa_pro", "languages": [], "modality": null, "name": "OfficeQA Pro", "openness": "unknown", "publisher": null, "released": "2025-06-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:office_qa_pro", "source_url": "https://github.com/OfficeQA/OfficeQA-Pro"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.2, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 66.2, "raw_min": 62.4, "source_reference": {"obs_id": "curated\u0000office_qa_pro\u0000tencent_hy4_preview\u0000office_qa_pro\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000office_qa_pro\u0000tencent_hy4_preview\u0000office_qa_pro\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "office_qa_pro", "source": "model_reports", "source_url": "https://github.com/OfficeQA/OfficeQA-Pro", "unit": "percent"}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OJBench is a competition-level code benchmark designed to assess the competitive-level code reasoning abilities of large language models. It comprises 232 programming competition problems from NOI and ICPC, categorized into Easy, Medium, and Hard difficulty levels. The benchmark evaluates models' ability to solve complex competitive programming challenges using Python and C++.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ojbench:kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ojbench", "languages": [], "modality": "text", "name": "OJBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.6, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.606, "raw_min": 0.271, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ojbench:kimi-k2.6", "reported_date": "2026-04-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500"}, "unit": null}, "slug": "llm-stats-ojbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OJBench (C++) is the C++ subset of OJBench, a competition-level code benchmark that evaluates large language models on programming competition problems from NOI and ICPC using C++ as the implementation language.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ojbench-cpp:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ojbench-cpp", "languages": [], "modality": "text", "name": "OJBench (C++)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.574, "raw_min": 0.574, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ojbench-cpp:kimi-k2.5", "reported_date": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500"}, "unit": null}, "slug": "llm-stats-ojbench-cpp", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "推理", "Reasoning", "数学", "Math", "大语言模型", "LLM", "逻辑推理", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A benchmark of 200 Olympiad math problems across algebra, geometry, number theory, and combinatorics. Available in English and Chinese, it features two difficulty levels: EASY (AIME-level) to test standard reasoning, and HARD to challenge advanced models. 一个包含 200 道奥林匹克数学题的基准测试，涵盖代数、几何、数论和组合。我们提供英文和中文版本，并提供两个难度等级：EASY（AIME水平）用于测试标准推理能力，以及 HARD 用于挑战高级模型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1705", "languages": ["English", "Chinese", "Multilingual"], "modality": null, "name": "OlymMATH", "openness": "open", "publisher": "中国人民大学", "released": "2025-03-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1705-olymmath", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OlymMATH", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "physics", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A challenging benchmark for promoting AGI with Olympiad-level bilingual multimodal scientific problems. Comprises 8,476 math and physics problems from international and Chinese Olympiads and the Chinese college entrance exam, featuring expert-level annotations for step-by-step reasoning. Includes both text-only and multimodal problems in English and Chinese.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:olympiadbench:qvq-72b-preview", "reported_at": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:olympiadbench", "languages": [], "modality": "multimodal", "name": "OlympiadBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 20.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.204, "raw_min": 0.204, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:olympiadbench:qvq-72b-preview", "reported_date": "2024-12-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500"}, "unit": null}, "slug": "llm-stats-olympiadbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "ACL 2024", "大语言模型", "LLM", "知识储备", "Knowledge", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OlympiadBench, an Olympiad-level bilingual multimodal scientific benchmark, featuring\n8,476 problems from Olympiad-level mathematics and physics competitions, including the\nChinese college entrance exam. Each problem is detailed with expert-level annotations\nfor step-by-step reasoning. OlympiadBench 是一个奥林匹克级别的双语多模态科学基准，包含来自奥林匹克级数学和物理竞赛的8,476道题目，包括中国高考。每道题目都配有专家级别的注释，提供逐步推理的详细说明。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1070", "languages": ["Chinese"], "modality": null, "name": "OlympiadBench", "openness": "open", "publisher": "OpenBMB", "released": "2024-06-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1070-olympiadbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OlympiadBench", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OlympicArena evaluates cognitive reasoning abilities. It includes 11,163 bilingual problems spanning seven fields and 62 international Olympic competitions. OlympicArena用于评估大模型的认知推理能力，包含来自7个领域、62项国际奥林匹克比赛的11163个双语问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1329", "languages": ["English", "Chinese"], "modality": null, "name": "OlympicArena", "openness": "restricted", "publisher": "Generative AI Research Lab", "released": "2024-06-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1329-olympicarena", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OlympicArena", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "数学", "Math", "大语言模型", "LLM", "逻辑推理", "Reasoning", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Omni-MATH focuses exclusively on mathematics and comprises a vast collection of 4428 competition-level problems with rigorous human annotation. These problems are meticulously categorized into over 33 sub-domains and span more than 10 distinct difficulty levels Omni-MATH用于评估LLM在奥林匹克水平上的数学推理能力，包括4428道竞赛级问题，并带有严格的人工注释。这些问题被精心分类为超过33个子领域，涵盖10多个不同的难度级别。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1244", "languages": [], "modality": null, "name": "Omni-MATH", "openness": "open", "publisher": "Peking University", "released": "2024-10-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1244-omni-math", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Omni-MATH", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OmniAlign-V datasets mainly focus on improving the alignment of Multi-modal Large Language Models(MLLMs) with human preference. It contains 205k high-quality Image-Quetion-Answer pairs with open-ended, creative quetions and long, knowledge-rich, comprehensive answers. OmniAlign-V 数据集主要关注提高多模态大型语言模型（MLLMs）与人类偏好的对齐。它包含 205k 个高质量的图像-问答对，包含开放式、创意性问题以及长篇、知识丰富、内容全面的答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1553", "languages": [], "modality": "multimodal", "name": "OmniAlign-V", "openness": "unknown", "publisher": "Shanghai Jiaotong University, Shanghai AI Laboratory, etc.", "released": "2025-02-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1553-omnialign-v", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniAlign-V", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A novel multimodal benchmark designed to evaluate large language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. Comprises 1,142 question-answer pairs covering 8 task categories from basic perception to complex inference, with a unique constraint that accurate responses require integrated understanding of all three modalities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omnibench:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omnibench", "languages": [], "modality": "multimodal", "name": "OmniBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.13, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.5613, "raw_min": 0.5613, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omnibench:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500"}, "unit": null}, "slug": "llm-stats-omnibench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OmniBench is a multi-dimensional benchmark for virtual agents, designed to systematically evaluate ten core capabilities such as planning, decision-making, and instruction comprehension through automatically generated task graphs with controllable complexity. OmniBench 是一个面向虚拟智能体的多维度评测基准，旨在通过自动化流程生成具有可控复杂度的任务图，系统评估智能体在计划、决策、指令理解等十个核心能力上的表现。该基准包含 36,000 个图结构任务，覆盖 20 个真实场景，并引入 OmniEval 框架，实现子任务级别的细粒度评估，显著提升了评测的效率和可扩展性。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1981", "languages": [], "modality": null, "name": "OmniBench", "openness": "unknown", "publisher": "Zhejiang University, Hangzhou, China , Ant Group, Hangzhou, China , etc.", "released": "2025-06-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1981-omnibench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Music component of OmniBench, a comprehensive benchmark for evaluating omni-language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. The music category includes various compositions and performances that require integrated understanding across text, image, and audio modalities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omnibench-music:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omnibench-music", "languages": [], "modality": "multimodal", "name": "OmniBench Music", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 52.83, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.5283, "raw_min": 0.5283, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omnibench-music:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500"}, "unit": null}, "slug": "llm-stats-omnibench-music", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "document_understanding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OmniDocBench evaluates multimodal models on document understanding tasks such as OCR, layout parsing, and structured document comprehension.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omnidocbench:mimo-v2.5", "reported_at": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omnidocbench", "languages": [], "modality": "multimodal", "name": "OmniDocBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.10000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.911, "raw_min": 0.872, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omnidocbench:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500"}, "unit": null}, "slug": "llm-stats-omnidocbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500", "unit": null}, {"aliases": ["OmniDocBench", "OmniDocBench1.5", "OmniDocBench 1.5"], "categories": ["multimodal"], "collected_at": null, "description": "Reported as an edit distance where lower is better, so it inverts the direction of every other row a card puts beside it.", "evidence_summary": {"document_count": 4, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000omnidocbench\u0000qwen3_5_model_card\u0000omnidocbench_1_5\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000omnidocbench\u0000qwen3_5_model_card\u0000omnidocbench_1_5\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:omnidocbench", "languages": [], "modality": null, "name": "OmniDocBench", "openness": "unknown", "publisher": null, "released": "2024-12-10", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:omnidocbench", "source_url": "https://github.com/opendatalab/OmniDocBench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.1, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 91.1, "raw_min": 90.8, "source_reference": {"obs_id": "curated\u0000omnidocbench\u0000moonshot_kimi_k3_model_card\u0000omnidocbench\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000omnidocbench\u0000moonshot_kimi_k3_model_card\u0000omnidocbench\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "omnidocbench", "source": "model_reports", "source_url": "https://github.com/opendatalab/OmniDocBench", "unit": "percent"}, {"aliases": [], "categories": ["多模态", "Multimodal", "长文本", "Long-Context", "Document content extraction", "多模态模型", "VLM", "长上下文", "Long Context", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OmniDocBench is a comprehensive benchmark for evaluating document parsing in real-world scenarios. It includes 981 PDF pages across 9 document types, annotated with dense paragraph-level bboxes with text and attributes. Along with its designed evaluation methods, it provides Fine-grained results. OmniDocBench是一个用于评估真实场景下多样性文档解析效果的评测集，它包含了981个页面，覆盖9种文档类型（包括研报、教材、报纸、手写笔记、杂志等），具有段落级别的位置标注和内容标注，还有阅读顺序标注和属性标注，并开发了配套的评测方法，使其既具备单模块的评测能力（包括布局检测，公式识别，表格识别，文本识别），又具备端到端的评测能力，针对不同元素提供了分页面以及分属性的精细化评测结果，精准定位模型文档解析的痛点问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1942", "languages": [], "modality": "multimodal", "name": "OmniDocBench", "openness": "unknown", "publisher": "Shanghai AI Laboratory, Abaka AI, 2077AI", "released": "2024-12-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1942-omnidocbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniDocBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "structured_output", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OmniDocBench 1.5 is a comprehensive benchmark for evaluating multimodal large language models on document understanding tasks, including OCR, document parsing, information extraction, and visual question answering across diverse document types. Lower Overall Edit Distance scores are better.", "evidence_summary": {"document_count": 1, "model_count": 18, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omnidocbench-1.5:gemini-3-pro-preview", "reported_at": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omnidocbench-1.5", "languages": [], "modality": "multimodal", "name": "OmniDocBench 1.5", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 18, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.60000000000001, "display_multiplier": 100, "model_count": 18, "model_count_basis": "source_model_id", "numeric_count": 18, "raw_max": 0.916, "raw_min": 0.115, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omnidocbench-1.5:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500"}, "unit": null}, "slug": "llm-stats-omnidocbench-1-5", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OmniGAIA evaluates multimodal perception and reasoning in agentic contexts, testing a model's ability to process diverse inputs and perform complex multi-step reasoning tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omnigaia:mimo-v2-omni", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omnigaia", "languages": [], "modality": "multimodal", "name": "OmniGAIA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 49.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.498, "raw_min": 0.498, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omnigaia:mimo-v2-omni", "reported_date": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500"}, "unit": null}, "slug": "llm-stats-omnigaia", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "代码", "Code", "智能体", "Agent", "Github Issue Resolution", "代码工程", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A multi-modal, multi-language benchmark dataset for the GitHub Issue Resolution task. 一个面向 GitHub Issue ResoLution任务的多语言、多模态基准数据集，包含以下特点: 1.支持 Python、Java、JS、TS 四种主流编程语言，2. 输入信息涵盖文本、图像、网页等多种模态，3. 提供可复现的 Docker 评估环境。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1801", "languages": ["Multilingual"], "modality": "multimodal", "name": "OmniGIRL", "openness": "restricted", "publisher": "Sun yat-sen University, Huawei Cloud Computing Technologies,Chongqing University", "released": "2025-05-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1801-omnigirl", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniGIRL", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A Universal Olympiad Level Mathematic Benchmark for Large Language Models containing 4,428 competition-level problems with rigorous human annotation, categorized into over 33 sub-domains and spanning more than 10 distinct difficulty levels", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omnimath:phi-4-reasoning", "reported_at": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omnimath", "languages": [], "modality": "text", "name": "OmniMath", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.89999999999999, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.819, "raw_min": 0.766, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omnimath:phi-4-reasoning-plus", "reported_date": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500"}, "unit": null}, "slug": "llm-stats-omnimath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "其他", "Other", "多模态模型", "VLM", "视频理解", "Video Understanding", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OmniMMI is a cutting-edge benchmark for multi-modal interaction, specifically designed for OmniLLMs in streaming video environments. It includes 1,121 videos and 2,290 questions, tackling the challenges of streaming video understanding and proactive reasoning across six unique subtasks. OmniMMI is a cutting-edge benchmark for multi-modal interaction, specifically designed for OmniLLMs in streaming video environments. It includes 1,121 videos and 2,290 questions, tackling the challenges of streaming video understanding and proactive reasoning across six unique subtasks.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2142", "languages": [], "modality": "multimodal", "name": "OmniMMI", "openness": "open", "publisher": "Beijing Institute of General Artificial Intelligence", "released": "2025-04-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2142-omnimmi", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniMMI", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "knowledge"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OmniScience is a broad scientific knowledge and reasoning benchmark that measures both answer accuracy and non-hallucination (calibrated abstention) across science domains.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omniscience:nemotron-3-ultra-550b-a55b", "reported_at": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omniscience", "languages": [], "modality": "text", "name": "OmniScience", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.787, "raw_min": 0.175, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omniscience:nemotron-3-ultra-550b-a55b", "reported_date": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500"}, "unit": null}, "slug": "llm-stats-omniscience", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "science", "knowledge"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OmniScience variant that reports the non-hallucination rate, defined as one minus the hallucination rate.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:omniscience-non-hallucination-rate:grok-4.5", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:omniscience-non-hallucination-rate", "languages": [], "modality": "text", "name": "OmniScience (non-hallucination rate)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 46.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.46, "raw_min": 0.46, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:omniscience-non-hallucination-rate:grok-4.5", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500"}, "unit": null}, "slug": "llm-stats-omniscience-non-hallucination-rate", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500", "unit": null}, {"aliases": ["OmniVideoBench", "Omni Video-Bench"], "categories": ["multimodal"], "collected_at": null, "description": "1,000 audio-visual reasoning questions over 628 videos (seconds to 30 minutes long), targeting synergistic audio-visual understanding in Omni MLLMs, by the NJU-LINK team; an ICLR 2026 poster. Video length and audio-video correlation vary widely, so per-model results depend heavily on which subset is sampled and on frame/audio chunking strategy.", "evidence_summary": {"document_count": null, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:omnivideobench", "languages": [], "modality": null, "name": "OmniVideoBench", "openness": "unknown", "publisher": null, "released": "2025-10-12", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:omnivideobench", "source_url": "https://arxiv.org/abs/2510.10689"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "omnivideobench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2510.10689", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OneMillion Bench evaluates AI agents on high-economic-value tasks that require sustained, reliable execution across long-horizon real-world workflows.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:onemillion-bench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:onemillion-bench", "languages": [], "modality": "text", "name": "OneMillion Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.8, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.688, "raw_min": 0.525, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:onemillion-bench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-onemillion-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500", "unit": null}, {"aliases": ["OneMillionBench", "One Million Bench"], "categories": ["long_context"], "collected_at": null, "description": "Long-context recall and use; a \"with tools\" figure is not comparable to a no-tools run.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000onemillionbench\u0000tencent_hy4_preview\u0000onemillionbench\u0000w/ tools\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000onemillionbench\u0000tencent_hy4_preview\u0000onemillionbench\u0000w/ tools\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:onemillionbench", "languages": [], "modality": null, "name": "OneMillionBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 65.4, "raw_min": 65.4, "source_reference": {"obs_id": "curated\u0000onemillionbench\u0000tencent_hy4_preview\u0000onemillionbench\u0000w/ tools\u0000Hy4 preview", "observation_id": "curated\u0000onemillionbench\u0000tencent_hy4_preview\u0000onemillionbench\u0000w/ tools\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "onemillionbench", "source": "model_reports", "source_url": "https://github.com/humanlaya/OneMillion-Bench", "unit": "percent"}, {"aliases": [], "categories": ["language", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OpenRewriteEval is a benchmark for evaluating open-ended rewriting of long-form texts, covering a wide variety of rewriting types expressed through natural language instructions including formality, expansion, conciseness, paraphrasing, and tone and style transfer.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:open-rewrite:llama-3.2-3b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:open-rewrite", "languages": [], "modality": "text", "name": "Open-rewrite", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 40.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.401, "raw_min": 0.401, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:open-rewrite:llama-3.2-3b-instruct", "reported_date": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500"}, "unit": null}, "slug": "llm-stats-open-rewrite", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "math", "physics", "psychology", "reasoning", "finance", "general", "healthcare", "chemistry", "economics"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "MMLU (Massive Multitask Language Understanding) is a comprehensive benchmark that measures a text model's multitask accuracy across 57 diverse academic and professional subjects. The test covers elementary mathematics, US history, computer science, law, morality, business ethics, clinical knowledge, and many other domains spanning STEM, humanities, social sciences, and professional fields. To attain high accuracy, models must possess extensive world knowledge and problem-solving ability.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openai-mmlu:gemma-3n-e2b-it", "reported_at": "2025-06-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openai-mmlu", "languages": [], "modality": "text", "name": "OpenAI MMLU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 35.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.356, "raw_min": 0.223, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openai-mmlu:gemma-3n-e4b-it", "reported_date": "2025-06-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500"}, "unit": null}, "slug": "llm-stats-openai-mmlu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Multi-round Co-reference Resolution (MRCR) benchmark for evaluating an LLM's ability to distinguish between multiple needles hidden in long context. Models are given a long, multi-turn synthetic conversation and must retrieve a specific instance of a repeated request, requiring reasoning and disambiguation skills beyond simple retrieval.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openai-mrcr:-2-needle-128k:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openai-mrcr:-2-needle-128k", "languages": [], "modality": "text", "name": "OpenAI-MRCR: 2 needle 128k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.19999999999999, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.952, "raw_min": 0.187, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openai-mrcr:-2-needle-128k:gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500"}, "unit": null}, "slug": "llm-stats-openai-mrcr-2-needle-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Multi-Round Co-reference Resolution benchmark that tests an LLM's ability to distinguish between multiple similar needles hidden in long conversations. Models must reproduce specific instances of content (e.g., 'Return the 2nd poem about tapirs') from multi-turn synthetic conversations, requiring reasoning about context, ordering, and subtle differences between similar outputs.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openai-mrcr:-2-needle-1m:gpt-4.1-2025-04-14", "reported_at": "2025-04-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openai-mrcr:-2-needle-1m", "languages": [], "modality": "text", "name": "OpenAI-MRCR: 2 needle 1M", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 58.599999999999994, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.586, "raw_min": 0.12, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openai-mrcr:-2-needle-1m:minimax-m1-40k", "reported_date": "2025-06-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500"}, "unit": null}, "slug": "llm-stats-openai-mrcr-2-needle-1m", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Multi-Round Co-reference Resolution (MRCR) benchmark that tests long-context reasoning by evaluating a model's ability to distinguish between similar outputs, reason about ordering, and reproduce specific content from multi-turn conversations containing multiple writing requests on overlapping topics at 256k tokens.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openai-mrcr:-2-needle-256k:gpt-5-2025-08-07", "reported_at": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openai-mrcr:-2-needle-256k", "languages": [], "modality": "text", "name": "OpenAI-MRCR: 2 needle 256k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.868, "raw_min": 0.868, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openai-mrcr:-2-needle-256k:gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500"}, "unit": null}, "slug": "llm-stats-openai-mrcr-2-needle-256k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OpenBookQA is a question-answering dataset modeled after open book exams for assessing human understanding. It contains 5,957 multiple-choice elementary-level science questions that probe understanding of 1,326 core science facts and their application to novel situations, requiring combination of open book facts with broad common knowledge through multi-hop reasoning.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openbookqa:mistral-nemo-instruct-2407", "reported_at": "2024-07-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openbookqa", "languages": [], "modality": "text", "name": "OpenBookQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.60000000000001, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.896, "raw_min": 0.494, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openbookqa:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500"}, "unit": null}, "slug": "llm-stats-openbookqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OpenBookQA contains questions that require multi-step reasoning, application of common-sense knowledge, and in-depth comprehension of text. It is a new type of question-answering dataset, modeled after open-book exams, designed to assess human understanding of a specific topic. OpenBookQA包含需要多步推理、运用常识知识、深入理解文本等能力的问题，是一种新型的问答数据集，其模式借鉴了开放式书本考试，用于评估人类对某一主题理解的程度。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:518", "languages": [], "modality": null, "name": "OpenbookQA", "openness": "unknown", "publisher": null, "released": "2018-09-08", "released_reference": {"basis": "paper_first_version", "note": "First version introducing OpenBookQA.", "source_key": "opencompass:518", "source_url": "https://arxiv.org/abs/1809.02789"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-518-openbookqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenbookQA", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OpenFinData is an open-source financial evaluation dataset jointly released by EastMoney and Shanghai Artificial Intelligence Laboratory. This dataset represents the most realistic industrial scenario requirements and is currently the most comprehensive and professional financial evaluation dataset. OpenFinData是由东方财富与上海人工智能实验室联合发布的开源金融评测数据集。该数据集代表了最真实的产业场景需求，是目前场景最全、专业性最深的金融评测数据集。它基于东方财富实际金融业务的多样化丰富场景，旨在为金融科技领域的研究者和开发者提供一个高质量的数据资源。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:631", "languages": [], "modality": null, "name": "OpenFinData", "openness": "restricted", "publisher": null, "released": "2023-12-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-631-openfindata", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenFinData", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OpenRCA is a benchmark for evaluating AI models on root cause analysis tasks. For each failure case, the model receives 1 point if all generated root-cause elements match the ground-truth ones, and 0 points if any mismatch is identified. The overall accuracy is the average score across all failure cases.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openrca:claude-opus-4-6", "reported_at": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openrca", "languages": [], "modality": "text", "name": "OpenRCA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 34.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.349, "raw_min": 0.349, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openrca:claude-opus-4-6", "reported_date": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500"}, "unit": null}, "slug": "llm-stats-openrca", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OpenTuringBench, a new benchmark based on OLLMs, designed to train and evaluate machine-generated text detectors on the Turing Test and Authorship Attribution problems. OpenTuringBench，这是一个基于OLLMs的新基准测试，旨在训练和评估机器生成文本检测器在图灵测试和作者归属问题上的性能。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1754", "languages": ["English"], "modality": null, "name": "OpenTuringBench", "openness": "unknown", "publisher": "DIMES Dept., University of Calabria", "released": "2025-04-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1754-openturingbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenTuringBench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OpenUnlearning is an efficient and modular benchmark platform designed to support and drive research on \"unlearning\" in large language models (LLMs). OpenUnlearning 是一个高效且模块化的基准平台，旨在支持和推动大型语言模型（LLM）中的“遗忘”（unlearning）研究。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1979", "languages": [], "modality": null, "name": "OpenUnlearning", "openness": "unknown", "publisher": "University of Massachusetts Amherst , Carnegie Mellon University ,etc.", "released": "2025-06-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1979-openunlearning", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenUnlearning", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OPT-BENCH is a comprehensive benchmark designed to evaluate large language model (LLM) agents on large-scale search space optimization problems, focusing on their iterative reasoning and problem-solving capabilities. OPT-BENCH 是一个面向大型语言模型（LLM）智能体的大规模搜索空间优化评测基准，旨在系统评估模型在迭代推理和解决复杂优化问题中的能力。 该基准包含 30 个任务，包括 20 个来自 Kaggle 的真实机器学习任务和 10 个经典 NP 问题，涵盖预测建模、图论和组合优化等领域。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1990", "languages": [], "modality": null, "name": "OPT-BENCH", "openness": "open", "publisher": "Tong Ji University , ShanghaiAILab , Nanjing University , Zhejiang University", "released": "2025-06-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1990-opt-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OPT-BENCH", "unit": null}, {"aliases": [], "categories": [], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:community:64d67847-06bd-423a-923c-c2acfab82281", "languages": [], "modality": null, "name": "OptimBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-community-64d67847-06bd-423a-923c-c2acfab82281", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A64d67847-06bd-423a-923c-c2acfab82281?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OR Bench is a large-scale benchmark designed to evaluate the excessive rejection behavior of large language models. OR-Bench 是一个旨在评估大型语言模型过度拒绝行为的大规模基准。它衡量LLM在过度拒绝和有害提示拒绝方面的表现，涵盖暴力、隐私等10个类别。基准包含8万个、1千个困难及6百个有害提示，通过Mixtral等工具自动化生成。它为未来安全对齐研究提供强大测试平台。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:2086", "languages": [], "modality": null, "name": "OR-Bench", "openness": "unknown", "publisher": "UCLA , UC Berkeley", "released": "2024-05-31", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the over-refusal benchmark.", "source_key": "opencompass:2086", "source_url": "https://arxiv.org/abs/2405.20947"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2086-or-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OR-Bench", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Orak (오락) is a foundational benchmark for evaluating Large Language Model (LLM) agents in diverse popular video games. Orak是一个基础性的基准,用于评估在各种流行视频游戏中的大型语言模型(LLM)代理。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1917", "languages": ["Korean"], "modality": null, "name": "Orak", "openness": "unknown", "publisher": "KRAFTON , SeoulNationalUniversity , NVIDIA , UniversityofWisconsin-Madison", "released": "2025-06-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1917-orak", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Orak", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "Memory Safety", "Open-Source Software", "Living Benchmark", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OSS-Bench replaces functions with LLM-generated code and evaluates them using three natural metrics: compilability, functional correctness, and memory safety, leveraging robust signals like compilation failures, test-suite violations, and sanitizer alerts as ground truth. OSS-Bench，这是一个基准生成器，它可以从真实的开源软件中自动构建大规模的实时评估任务。OSS-Bench 将函数替换为 LLM 生成的代码，并使用三个自然指标（可编译性、功能正确性和内存安全性）对其进行评估，并利用编译失败、测试套件违规和Sanitizer警报等稳健信号作为基准事实。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2005", "languages": [], "modality": null, "name": "OSS-Bench", "openness": "unknown", "publisher": "National University of Singapore", "released": "2025-06-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2005-oss-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OSS-Bench", "unit": null}, {"aliases": [], "categories": ["multimodal", "general", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OSWorld: The first-of-its-kind scalable, real computer environment for multimodal agents, supporting task setup, execution-based evaluation, and interactive learning across Ubuntu, Windows, and macOS with 369 computer tasks involving real web and desktop applications, OS file I/O, and multi-application workflows", "evidence_summary": {"document_count": 1, "model_count": 20, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:osworld:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:osworld", "languages": [], "modality": "multimodal", "name": "OSWorld", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 20, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.8, "display_multiplier": 100, "model_count": 20, "model_count_basis": "source_model_id", "numeric_count": 20, "raw_max": 0.788, "raw_min": 0.0592, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:osworld:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500"}, "unit": null}, "slug": "llm-stats-osworld", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500", "unit": null}, {"aliases": ["OSWorld", "OSWorld-Verified", "OSWorld 2.0"], "categories": ["computer_use"], "collected_at": null, "description": "Screen resolution, OS image and action space all change results. The Verified and 2.0 revisions are not comparable to the original.", "evidence_summary": {"document_count": 5, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000osworld\u0000qwen3_5_model_card\u0000osworld_verified\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000osworld\u0000qwen3_5_model_card\u0000osworld_verified\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:osworld", "languages": [], "modality": null, "name": "OSWorld", "openness": "unknown", "publisher": null, "released": "2024-04-11", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:osworld", "source_url": "https://os-world.github.io/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.8, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 84.8, "raw_min": 58.3, "source_reference": {"obs_id": "curated\u0000osworld\u0000moonshot_kimi_k3_model_card\u0000osworld_verified\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000osworld\u0000moonshot_kimi_k3_model_card\u0000osworld_verified\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "osworld", "source": "model_reports", "source_url": "https://os-world.github.io/", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "general", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OSWorld 2.0 is a benchmark of 108 long-horizon, real-world computer-use workflows spanning everyday and professional tasks. Each task is an end-to-end workflow that takes human users a median of about 1.6 hours, scored with a binary-completion metric, and targets challenges such as dynamic environments, cross-source reasoning, and implicit-state inference.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:osworld-2.0:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:osworld-2.0", "languages": [], "modality": "multimodal", "name": "OSWorld 2.0", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.6, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.706, "raw_min": 0.456, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:osworld-2.0:claude-opus-5", "reported_date": "2026-07-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500"}, "unit": null}, "slug": "llm-stats-osworld-2-0", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OSWorld is a scalable, real computer environment benchmark for evaluating multimodal agents on open-ended tasks across Ubuntu, Windows, and macOS. It comprises 369 computer tasks involving real web and desktop applications, OS file I/O, and multi-application workflows. The benchmark evaluates agents' ability to interact with computer interfaces using screenshots and actions in realistic computing environments.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:osworld-extended:claude-3-5-sonnet-20241022", "reported_at": "2024-10-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:osworld-extended", "languages": [], "modality": "multimodal", "name": "OSWorld Extended", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 22.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.22, "raw_min": 0.22, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:osworld-extended:claude-3-5-sonnet-20241022", "reported_date": "2024-10-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500"}, "unit": null}, "slug": "llm-stats-osworld-extended", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "general", "grounding", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OSWorld Screenshot-only: A variant of the OSWorld benchmark that evaluates multimodal AI agents using only screenshot observations to complete open-ended computer tasks across real operating systems (Ubuntu, Windows, macOS). Tests agents' ability to perform complex workflows involving web apps, desktop applications, file I/O, and multi-application tasks through visual interface understanding and GUI grounding.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:osworld-screenshot-only:claude-3-5-sonnet-20241022", "reported_at": "2024-10-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:osworld-screenshot-only", "languages": [], "modality": "multimodal", "name": "OSWorld Screenshot-only", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 14.899999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.149, "raw_min": 0.149, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:osworld-screenshot-only:claude-3-5-sonnet-20241022", "reported_date": "2024-10-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500"}, "unit": null}, "slug": "llm-stats-osworld-screenshot-only", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "grounding", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OSWorld-G (Grounding) evaluates screenshot grounding accuracy for OS automation tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:osworld-g:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:osworld-g", "languages": [], "modality": "image", "name": "OSWorld-G", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.683, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.683, "raw_min": 0.683, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:osworld-g:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500"}, "unit": null}, "slug": "llm-stats-osworld-g", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "general", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OSWorld-Verified is a verified subset of OSWorld, a scalable real computer environment for multimodal agents supporting task setup, execution-based evaluation, and interactive learning across Ubuntu, Windows, and macOS.", "evidence_summary": {"document_count": 1, "model_count": 24, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:osworld-verified:gpt-5.3-codex", "reported_at": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:osworld-verified", "languages": [], "modality": "multimodal", "name": "OSWorld-Verified", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 24, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.1, "display_multiplier": 100, "model_count": 24, "model_count_basis": "source_model_id", "numeric_count": 24, "raw_max": 0.861, "raw_min": 0.39, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:osworld-verified:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500"}, "unit": null}, "slug": "llm-stats-osworld-verified", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OVBench is an online video understanding benchmark that evaluates a model's ability to perceive, memorize, and reason about real-time video streams as they unfold.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ovbench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ovbench", "languages": [], "modality": "multimodal", "name": "OVBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.7, "raw_min": 0.697, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ovbench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500"}, "unit": null}, "slug": "llm-stats-ovbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "OVOBench (Online Video Online Benchmark) evaluates streaming video understanding, testing a model's ability to perceive and respond to video content in real time as it unfolds.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ovobench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ovobench", "languages": [], "modality": "multimodal", "name": "OVOBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.7, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.807, "raw_min": 0.792, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ovobench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500"}, "unit": null}, "slug": "llm-stats-ovobench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "P-MMEval is a comprehensive multilingual multitask benchmark, covering effective fundamental and capability-specialized datasets. P-MMEval是一个全面的多语言多任务基准，涵盖了高效的基础和专项能力数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1283", "languages": ["Multilingual"], "modality": null, "name": "P-MMEval", "openness": "unknown", "publisher": "Tongyi Lab, Alibaba Group Inc", "released": "2024-11-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1283-p-mmeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/P-MMEval", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PaperBench is a benchmark for evaluating AI agents on their ability to replicate research papers. It tests models on complex, multi-step workflows involving code implementation, experimentation, and reproducing scientific results from academic publications.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:paperbench:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:paperbench", "languages": [], "modality": "text", "name": "PaperBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.93, "raw_min": 0.526, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:paperbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500"}, "unit": null}, "slug": "llm-stats-paperbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "智能体", "Agent", "任务执行", "Task Execution", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PaperBench is a benchmark evaluating the ability of AI agents to replicate state-of-the-art AI research. PaperBench，这是一个评估AI代理复制最新AI研究能力的基准测试。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1730", "languages": [], "modality": null, "name": "PaperBench", "openness": "unknown", "publisher": "OpenAI", "released": "2025-04-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1730-paperbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PaperBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "语言", "Language", "OCR", "NLP", "Computer Vision", "多模态模型", "VLM", "图像理解", "Image Understanding", "语言理解", "Comprehension", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PsOCR is a large-scale synthetic dataset for Optical Character Recognition in low-resource Pashto language. PsOCR is a large-scale synthetic dataset for Optical Character Recognition in low-resource Pashto language.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1836", "languages": ["Arabic"], "modality": "multimodal", "name": "PashtoOCR", "openness": "unknown", "publisher": "Zirak AI", "released": "2025-05-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1836-pashtoocr", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PashtoOCR", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PathMMU is a massive multimodal expert-level benchmark for understanding and reasoning in pathology, containing 33,428 multimodal multi-choice questions and 24,067 images validated by seven pathologists. It evaluates Large Multimodal Models (LMMs) performance on pathology tasks, with the top-performing model GPT-4V achieving only 49.8% zero-shot performance compared to 71.8% for human pathologists.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:pathmcqa:medgemma-4b-it", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:pathmcqa", "languages": [], "modality": "multimodal", "name": "PathMCQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 69.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.698, "raw_min": 0.698, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:pathmcqa:medgemma-4b-it", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500"}, "unit": null}, "slug": "llm-stats-pathmcqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "ACL 2024", "物理智能", "Embodied AI", "具身交互", "Embodied Interaction", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PCA-Bench is a multimodal decisionmaking benchmark for evaluating the integrated capabilities of Multimodal Large Language Models (MLLMs). Departing from previous\nbenchmarks focusing on simplistic tasks and individual model capability. PCA-Bench 是一个多模态决策基准，用于评估多模态大型语言模型（MLLMs）的综合能力。与之前专注于简单任务和单个模型能力的基准不同，PCA-Bench 引入了三个复杂场景：自动驾驶、家庭机器人和开放世界游戏。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1076", "languages": [], "modality": null, "name": "PCA-Bench", "openness": "open", "publisher": "Alibaba", "released": "2024-02-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1076-pca-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PCA-Bench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PerceptionBench is Moonshot AI's internal benchmark for evaluating atomic visual perception capabilities.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:perceptionbench:kimi-k3", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:perceptionbench", "languages": [], "modality": "multimodal", "name": "PerceptionBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.635, "raw_min": 0.585, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:perceptionbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500"}, "unit": null}, "slug": "llm-stats-perceptionbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "physics", "reasoning", "spatial_reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A novel multimodal video benchmark designed to evaluate perception and reasoning skills of pre-trained models across video, audio, and text modalities. Contains 11.6k real-world videos (average 23 seconds) filmed by participants worldwide, densely annotated with six types of labels. Focuses on skills (Memory, Abstraction, Physics, Semantics) and reasoning types (descriptive, explanatory, predictive, counterfactual). Shows significant performance gap between human baseline (91.4%) and state-of-the-art video QA models (46.2%).", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:perceptiontest:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:perceptiontest", "languages": [], "modality": "multimodal", "name": "PerceptionTest", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.732, "raw_min": 0.705, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:perceptiontest:qwen2.5-vl-72b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500"}, "unit": null}, "slug": "llm-stats-perceptiontest", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "智能体", "Agent", "AI Assistant", "personalization", "任务执行", "Task Execution", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PersonaLens  a large-scale benchmark specifically designed to evaluate personalization in task-oriented dialogues. The benchmark features 1,500 in-depth user profiles, each integrating real demographic data, detailed cross-domain preferences, and rich interaction histories. PersonaLens是一个专为任务导向型对话设计的、大规模的个性化能力评测基准。它包含1,500个深度用户画像，每个画像都集成了真实的人口统计信息、详尽的个人偏好及历史互动记录。这些画像与覆盖20个领域的111项真实世界任务相结合，并辅以动态的“情景上下文”来模拟现实世界的复杂性。为了实现自动化、可扩展的评估，该基准引入了两个LLM驱动的智能体：一个“用户智能体”负责模拟真人与AI进行对话，另一个“评判智能体”则对个性化水平、任务成功率和对话质量进行系统性打分。PersonaLens旨在为研究社区提供一个强大可靠的工具，共同推动下一代更懂你、更智能的AI助手的研发。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1959", "languages": [], "modality": null, "name": "PersonaLens", "openness": "unknown", "publisher": "University of Edinburgh, Amazon, UCL", "released": "2025-06-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1959-personalens", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PersonaLens", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PertEval-scFM is a standardized framework to evaluate single-cell foundation models (scFMs) for predicting cellular perturbation effects. PertEval-scFM 是一个评估单细胞基础模型（scFM）扰动效应预测的标准化框架。它从分布偏移泛化性、扰动强度及上下文对齐三方面评测。测试集包含Norman、Replogle等数据集，涵盖数千扰动样本和基因。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:2087", "languages": [], "modality": null, "name": "PertEval-scFM", "openness": "unknown", "publisher": "Queen Mary University of London , University of Oxford , etc.", "released": "2024-10-03", "released_reference": {"basis": "paper_first_version", "note": "The first preprint was posted October 3; the October 2 date in the DOI is not its posting date.", "source_key": "opencompass:2087", "source_url": "https://www.biorxiv.org/content/10.1101/2024.10.02.616248v1"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2087-perteval-scfm", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PertEval-scFM", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PhiBench is an internal benchmark designed to evaluate diverse skills and reasoning abilities of language models, covering a wide range of tasks including coding (debugging, extending incomplete code, explaining code snippets) and mathematics (identifying proof errors, generating related problems). Created by Microsoft's research team to address limitations of standard academic benchmarks and guide the development of the Phi-4 model.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:phibench:phi-4", "reported_at": "2024-12-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:phibench", "languages": [], "modality": "text", "name": "PhiBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.742, "raw_min": 0.562, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:phibench:phi-4-reasoning-plus", "reported_date": "2025-04-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500"}, "unit": null}, "slug": "llm-stats-phibench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500", "unit": null}, {"aliases": [], "categories": ["physics", "reasoning", "science"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PHYBench is a benchmark of real-world physics problems spanning mechanics, electromagnetism, thermodynamics, optics, and modern physics, designed to evaluate physical perception and multi-step quantitative reasoning in large language models.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:phybench:hy3", "reported_at": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:phybench", "languages": [], "modality": "text", "name": "PHYBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.774, "raw_min": 0.774, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:phybench:hy3", "reported_date": "2026-07-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500"}, "unit": null}, "slug": "llm-stats-phybench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "物理智能", "Embodied AI", "跨模态推理", "Cross-modal Reasoning", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PhyGenBench is a benchmark designed to evaluate physical commonsense correctness in text-to-video generation models. I PhyGenBench 是一个评估T2V模型物理常识的基准。它涵盖力学、光学、热学和材料特性四大领域27种物理定律，包含160个提示。结合PhyGenEval框架，利用视听和大语言模型自动化评估模型的物理理解能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2080", "languages": [], "modality": "multimodal", "name": "PhyGenBench", "openness": "unknown", "publisher": "Shanghai Jiao Tong University , Shanghai AI Laboratory ,etc.", "released": "2024-10-07", "released_reference": {"basis": "paper_first_version", "note": "The paper introducing PhyGenBench's 160 physical-commonsense video prompts.", "source_key": "opencompass:2080", "source_url": "https://arxiv.org/abs/2410.05363"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2080-phygenbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PhyGenBench", "unit": null}, {"aliases": [], "categories": ["math", "physics", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PHYSICS is a comprehensive benchmark for university-level physics problem solving, containing 1,297 expert-annotated problems covering six core areas: classical mechanics, quantum mechanics, thermodynamics and statistical mechanics, electromagnetism, atomic physics, and optics. Each problem requires advanced physics knowledge and mathematical reasoning. Even advanced models like o3-mini achieve only 59.9% accuracy.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:physicsfinals:gemini-1.5-flash", "reported_at": "2024-05-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:physicsfinals", "languages": [], "modality": "text", "name": "PhysicsFinals", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.9, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.639, "raw_min": 0.574, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:physicsfinals:gemini-1.5-pro", "reported_date": "2024-05-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500"}, "unit": null}, "slug": "llm-stats-physicsfinals", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "推理", "Reasoning", "科学智能", "AI for Science", "知识储备", "Knowledge", "逻辑推理", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PhysReason is a comprehensive physics-based reasoning benchmark consisting of 1,200 physics problems spanning multiple domains, with a focus on both knowledge-based (25%) and reasoning-based (75%) questions. PhysReason 是一个包含 1,200 个物理问题的综合物理推理基准，涵盖多个领域，重点关注基于知识（25%）和推理（75%）的问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1576", "languages": [], "modality": null, "name": "PhysReason", "openness": "unknown", "publisher": "Xi’an Jiaotong University, etc.", "released": "2025-02-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1576-physreason", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PhysReason", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "指令跟随", "Instruct", "创作", "Creation", "Image Editing", "AIGC", "Physics", "物理智能", "Embodied AI", "指令遵循", "Instruction Following", "语言生成", "Generation", "视觉生成", "Visual Generation", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2350", "languages": [], "modality": "multimodal", "name": "PICABench", "openness": "open", "publisher": "Shanghai AI Laboratory", "released": "2025-10-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2350-picabench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PICABench", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PinchBench evaluates coding agents on real-world agentic coding tasks, measuring both best-case and average performance across complex software engineering scenarios.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:pinchbench:mimo-v2-omni", "reported_at": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:pinchbench", "languages": [], "modality": "text", "name": "PinchBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.0, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.9, "raw_min": 0.6822, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:pinchbench:nemotron-3-ultra-550b-a55b", "reported_date": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500"}, "unit": null}, "slug": "llm-stats-pinchbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["physics", "reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PIQA (Physical Interaction: Question Answering) is a benchmark dataset for physical commonsense reasoning in natural language. It tests AI systems' ability to answer questions requiring physical world knowledge through multiple choice questions with everyday situations, focusing on atypical solutions inspired by instructables.com. The dataset contains 21,000 multiple choice questions where models must choose the most appropriate solution for physical interactions.", "evidence_summary": {"document_count": 1, "model_count": 11, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:piqa:gemma-2-27b-it", "reported_at": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:piqa", "languages": [], "modality": "text", "name": "PIQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 11, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.6, "display_multiplier": 100, "model_count": 11, "model_count_basis": "source_model_id", "numeric_count": 11, "raw_max": 0.886, "raw_min": 0.552, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:piqa:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500"}, "unit": null}, "slug": "llm-stats-piqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PIQA is a physical interaction question answering task, which requires to select the most reasonable solution based on the given scenario and two possible solutions. This task is designed to test the model's knowledge in physical commonsense. This dataset consists of 16k training samples, 800 development samples and 2k test samples, all on English text. PIQA是一个物理交互问答任务，要求根据给定的场景和两个可能的解决方案，选择最合理的方案。这个任务是为了测试模型在物理常识方面的知识。这个数据集包含了16000个训练样本，800个开发样本和2000个测试样本，所有的文本都是英文文本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:532", "languages": ["English"], "modality": null, "name": "PIQA", "openness": "unknown", "publisher": null, "released": "2019-11-26", "released_reference": {"basis": "paper_first_version", "note": "The physical-commonsense question-answering paper, not the unrelated image-quality library.", "source_key": "opencompass:532", "source_url": "https://arxiv.org/abs/1911.11641"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-532-piqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PIQA", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "大语言模型", "LLM", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PlanBench is an extensible benchmark suite based on the kinds of domains used in the automated planning community, especially in the International Planning Competition, to test the capabilities of LLMs in planning or reasoning about actions and change. PlanBench用于评估LLM的规划能力，基于自动化规划社区（尤其是在国际规划竞赛）中涉及的各种领域来测试大模型在规划或推理行动和变更方面的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1251", "languages": [], "modality": null, "name": "PlanBench", "openness": "unknown", "publisher": "Arizona State University", "released": "2022-06-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1251-planbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PlanBench", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "knowledge"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PLawBench evaluates language models on professional legal knowledge and reasoning tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:plawbench:qwen3.8-max", "reported_at": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:plawbench", "languages": [], "modality": "text", "name": "PLawBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.2, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.732, "raw_min": 0.732, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:plawbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500"}, "unit": null}, "slug": "llm-stats-plawbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A medical visual question answering benchmark built on biomedical literature and medical figures.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:pmc-vqa:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:pmc-vqa", "languages": [], "modality": "multimodal", "name": "PMC-VQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.3, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.633, "raw_min": 0.62, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:pmc-vqa:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500"}, "unit": null}, "slug": "llm-stats-pmc-vqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "spatial_reasoning", "grounding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PointArena is a comprehensive platform for evaluating multimodal pointing across diverse reasoning scenarios. It includes Point-Bench, a curated dataset of ~1,000 pointing tasks across five categories: Spatial (positional references), Affordance (functional part identification), Counting (attribute-based grouping), Steerable (relative pointing), and Reasoning (open-ended visual inference). The benchmark evaluates language-guided pointing capabilities in vision-language models.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:pointgrounding:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:pointgrounding", "languages": [], "modality": "multimodal", "name": "PointGrounding", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.665, "raw_min": 0.665, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:pointgrounding:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500"}, "unit": null}, "slug": "llm-stats-pointgrounding", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PokerBench contains natural language game scenarios and optimal decisions computed by solvers in No Limit Texas Hold’em. It is divided into pre-flop and post-flop datasets, each with training and test splits. PokerBench包含自然语言游戏场景和由求解器在无限制德州扑克中计算出的最优决策。它分为前注和后注数据集，每个数据集都包含训练集和测试集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1675", "languages": [], "modality": null, "name": "PokerBench", "openness": "open", "publisher": "University of California, Berkeley, etc.", "released": "2025-01-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1675-pokerbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PokerBench", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Polymath is a challenging multi-modal mathematical reasoning benchmark designed to evaluate the general cognitive reasoning abilities of Multi-modal Large Language Models (MLLMs). The benchmark comprises 5,000 manually collected high-quality images of cognitive textual and visual challenges across 10 distinct categories, including pattern recognition, spatial reasoning, and relative reasoning.", "evidence_summary": {"document_count": 1, "model_count": 23, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:polymath:qwen3-235b-a22b-instruct-2507", "reported_at": "2025-07-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:polymath", "languages": [], "modality": "multimodal", "name": "PolyMATH", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 23, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.5, "display_multiplier": 100, "model_count": 23, "model_count_basis": "source_model_id", "numeric_count": 23, "raw_max": 0.865, "raw_min": 0.082, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:polymath:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500"}, "unit": null}, "slug": "llm-stats-polymath", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PolyMath is a multilingual mathematical reasoning benchmark covering 18 languages and 4 difficulty levels from easy to hard, ensuring difficulty comprehensiveness, language diversity, and high-quality translation. The benchmark evaluates mathematical reasoning capabilities of large language models across diverse linguistic contexts, making it a highly discriminative multilingual mathematical benchmark.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:polymath-en:kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:polymath-en", "languages": [], "modality": "text", "name": "PolyMath-en", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.10000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.651, "raw_min": 0.651, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:polymath-en:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500"}, "unit": null}, "slug": "llm-stats-polymath-en", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "safety", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Polling-based Object Probing Evaluation (POPE) is a benchmark for evaluating object hallucination in Large Vision-Language Models (LVLMs). POPE addresses the problem where LVLMs generate objects inconsistent with target images by using a polling-based query method that asks yes/no questions about object presence in images, providing more stable and flexible evaluation of object hallucination.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:pope:phi-3.5-vision-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:pope", "languages": [], "modality": "multimodal", "name": "POPE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.887, "raw_min": 0.856, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:pope:lfm-2.5-vl-3b", "reported_date": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500"}, "unit": null}, "slug": "llm-stats-pope", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "POPE is an improved evaluation method for LVLMs' object hallucination by proposing a polling-based query method. It offers a more stable and flexible solution. POPE用于评估视觉语言模型的物体幻觉，基于轮询的查询方法设计，提供了一种更稳定、更灵活的评估方案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1362", "languages": [], "modality": "multimodal", "name": "POPE", "openness": "restricted", "publisher": "Renmin University of China", "released": "2023-05-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1362-pope", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/POPE", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PopQA is an entity-centric open-domain question-answering dataset consisting of 14,000 QA pairs designed to evaluate language models' ability to memorize and recall factual knowledge across entities with varying popularity levels. The dataset probes both parametric memory (stored in model parameters) and non-parametric memory effectiveness, with questions covering 16 diverse relationship types from Wikidata converted to natural language using templates. Created by sampling knowledge triples from Wikidata and converting them to natural language questions, focusing on long-tail entities to understand LMs' strengths and limitations in memorizing factual knowledge.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:popqa:granite-3.3-8b-base", "reported_at": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:popqa", "languages": [], "modality": "text", "name": "PopQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 26.169999999999998, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.2617, "raw_min": 0.229, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:popqa:granite-3.3-8b-base", "reported_date": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500"}, "unit": null}, "slug": "llm-stats-popqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The PosterSum dataset is a multimodal benchmark designed for the summarization of scientific posters into research paper abstracts. The dataset consists of 16,305 research posters collected from major machine learning conferences, including ICLR, ICML, and NeurIPS, spanning the years 2022-2024. PosterSum 数据集是一个多模态基准数据集，旨在将科学海报总结成研究论文摘要。该数据集包含从2022-2024年的主要机器学习会议（包括 ICLR、ICML 和 NeurIPS）收集的 16,305 篇研究海报。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1579", "languages": [], "modality": "multimodal", "name": "PosterSum", "openness": "unknown", "publisher": "University of Edinburgh", "released": "2025-02-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1579-postersum", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PosterSum", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code", "systems"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PostTrainBench evaluates a model's ability to autonomously post-train base models. Given pretrain-only base models, the agent must complete the full pipeline of data synthesis, training, evaluation, and iteration within a time budget, scored across downstream benchmarks such as AIME2025, BFCL, GPQA Main, GSM8K, and HumanEval.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:posttrainbench:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:posttrainbench", "languages": [], "modality": "text", "name": "PostTrainBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 39.800000000000004, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.398, "raw_min": 0.165, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:posttrainbench:glm-5.3", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500"}, "unit": null}, "slug": "llm-stats-posttrainbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500", "unit": null}, {"aliases": ["PostTrainBench", "PostTrainBench Lite"], "categories": ["ai_research"], "collected_at": null, "description": "Measures a model's ability to run post-training itself. Hardware differs between reported runs (H100 vs H20), which moves the score.", "evidence_summary": {"document_count": 4, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000posttrainbench\u0000moonshot_kimi_k3_model_card\u0000posttrainbench\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "first_score_reported_at": "2026-06-13", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000posttrainbench\u0000moonshot_kimi_k3_model_card\u0000posttrainbench\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:posttrainbench", "languages": [], "modality": null, "name": "PostTrainBench", "openness": "unknown", "publisher": null, "released": "2026-01-20", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:posttrainbench", "source_url": "https://posttrainbench.com/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 36.6, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 36.6, "raw_min": 34.3, "source_reference": {"obs_id": "curated\u0000posttrainbench\u0000moonshot_kimi_k3_model_card\u0000posttrainbench\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000posttrainbench\u0000moonshot_kimi_k3_model_card\u0000posttrainbench\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "posttrainbench", "source": "model_reports", "source_url": "https://posttrainbench.com/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "agents", "code", "systems"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PostTrainBench Lite measures whether an agent can design and execute a full post-training strategy (data, prompts, RL recipe, and eval loop) for a pretrained base model under a constrained time budget, scored as normalized mean reward over the improvement window.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:posttrainbench-lite:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:posttrainbench-lite", "languages": [], "modality": "text", "name": "PostTrainBench Lite", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.5, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.515, "raw_min": 0.296, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:posttrainbench-lite:gpt-5.6-terra", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500"}, "unit": null}, "slug": "llm-stats-posttrainbench-lite", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "knowledge", "finance"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PRBench-Finance evaluates professional reasoning on finance tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:prbench-finance:qwen3.8-max", "reported_at": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:prbench-finance", "languages": [], "modality": "text", "name": "PRBench-Finance", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 58.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.583, "raw_min": 0.583, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:prbench-finance:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500"}, "unit": null}, "slug": "llm-stats-prbench-finance", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "knowledge"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PRBench-Legal evaluates professional reasoning on legal tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:prbench-legal:qwen3.8-max", "reported_at": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:prbench-legal", "languages": [], "modality": "text", "name": "PRBench-Legal", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.599999999999994, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.576, "raw_min": 0.576, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:prbench-legal:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500"}, "unit": null}, "slug": "llm-stats-prbench-legal", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "PresentBench evaluates AI agents on producing presentation-style deliverables, such as generating lesson-plan slides and structured documents from source materials.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:presentbench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:presentbench", "languages": [], "modality": "text", "name": "PresentBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 54.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.546, "raw_min": 0.483, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:presentbench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500"}, "unit": null}, "slug": "llm-stats-presentbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "数理能力", "Math", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "PRMBench is a benchmark dataset for evaluating process-level reward models (PRMs). It consists of 6,216 data instances, each containing a question, a solution process, and a modified process with errors. PRMBench 是一个用于评估过程级奖励模型（PRM）的基准数据集。它包含 6,216 个数据实例，每个实例包含一个问题、一个解决方案过程以及一个包含错误的修改过程。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1681", "languages": [], "modality": null, "name": "PRMBench_Preview", "openness": "open", "publisher": "FDU, etc.", "released": "2025-01-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1681-prmbench-preview", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PRMBench_Preview", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ProBench is a benchmark that contains open-ended multimodal queries that require intensive expert-level knowledge to solve. ProBench是一个包含需要大量专家级知识来解决的开放式多模态查询的基准。ProBench 包含 10 个任务领域和 56 个子领域，支持 17 种语言，并支持最多 13 轮对话。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1632", "languages": [], "modality": "multimodal", "name": "ProBench", "openness": "restricted", "publisher": "ANU, NTU", "released": "2025-03-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1632-probench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProBench", "unit": null}, {"aliases": [], "categories": ["学科", "Examination", "推理", "Reasoning", "大语言模型", "LLM", "数理能力", "Math", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ProcessBench can measure the ability to identify erroneous steps in mathematical reasoning. It consists of 3,400 test cases, primarily focused on competition- and Olympiad-level math problems. ProcessBench，用于衡量识别数学推理中错误步骤的能力。它包含 3,400 个测试案例，主要关注竞赛和奥林匹克级别的数学问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1621", "languages": [], "modality": null, "name": "ProcessBench", "openness": "open", "publisher": "QwenTeam, Alibaba Inc.", "released": "2024-12-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1621-processbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProcessBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "knowledge", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ProfBench evaluates models on professional-domain reasoning and knowledge-work tasks, including search-augmented question answering across expert fields.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:profbench:nemotron-3-ultra-550b-a55b", "reported_at": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:profbench", "languages": [], "modality": "text", "name": "ProfBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 56.00000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.56, "raw_min": 0.56, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:profbench:nemotron-3-ultra-550b-a55b", "reported_date": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500"}, "unit": null}, "slug": "llm-stats-profbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Program Bench evaluates code-generation agents by asking them to recreate a program's behavior from only a compiled binary and documentation. It spans 200 tasks from small CLI tools to large systems such as FFmpeg and SQLite, with submissions judged against more than 248,000 fuzz-generated behavioral tests.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:program-bench:kimi-k2.7-code", "reported_at": "2026-06-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:program-bench", "languages": [], "modality": "text", "name": "Program Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.8, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.778, "raw_min": 0.19, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:program-bench:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-program-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["coding"], "collected_at": null, "description": "Cleanroom program-rebuild tasks; most models score low, so small differences sit within run-to-run noise.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000programbench\u0000tencent_hy4_preview\u0000programbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000programbench\u0000tencent_hy4_preview\u0000programbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:programbench", "languages": [], "modality": null, "name": "ProgramBench", "openness": "unknown", "publisher": null, "released": "2026-05-05", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:programbench", "source_url": "https://arxiv.org/abs/2605.03546"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 17.5, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 17.5, "raw_min": 17.5, "source_reference": {"obs_id": "curated\u0000programbench\u0000tencent_hy4_preview\u0000programbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000programbench\u0000tencent_hy4_preview\u0000programbench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "programbench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2605.03546", "unit": "percent"}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ProJudge is a comprehensive, multi-modal, multi-discipline, and multi-difficulty benchmark specifically designed for evaluating abilities of MLLM-based process judges.It comprises 2,400 test cases and 50,118 step-level labels, spanning four scientific disciplines with diverse difficulty levels and ProJudge 是一个针对基于 MLLM 的过程裁判能力的全面、多模态、多学科和多难度的基准。它包含 2,400 个测试案例和 50,118 个步骤级标签，涵盖四个科学学科，难度级别和内容多样化。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1633", "languages": [], "modality": "multimodal", "name": "ProJudge", "openness": "restricted", "publisher": "WHU, Shanghai AI Laboratory, etc.", "released": "2025-03-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1633-projudge", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProJudge", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "科学智能", "AI for Science", "数理能力", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ProofNet is a benchmark for autoformalization and formal proving of undergraduate-level mathematics. The ProofNet benchmarks consists of 371 examples, each consisting of a formal theorem statement in Lean 3, a natural language theorem statement, and a natural language proof. ProofNet 是一个用于本科数学的自动形式化和形式证明的基准。包含 371 个示例，每个示例包括一个 Lean 3 中的形式定理陈述、一个自然语言定理陈述和一个自然语言证明。这些问题主要来自流行的本科纯数学教材，涵盖实分析、复分析、线性代数、抽象代数和拓扑等主题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1120", "languages": [], "modality": null, "name": "ProofNet", "openness": "open", "publisher": "Yale University & University of Warsaw & Carnegie Mellon University", "released": "2023-02-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1120-proofnet", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProofNet", "unit": null}, {"aliases": [], "categories": ["safety", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ProtocolQA is a multiple-choice benchmark on troubleshooting failed experimental outcomes from common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:protocolqa:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:protocolqa", "languages": [], "modality": "text", "name": "ProtocolQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 79.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.79, "raw_min": 0.79, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:protocolqa:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500"}, "unit": null}, "slug": "llm-stats-protocolqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Q-Bench/Q-Bench+ is a benchmark for general-purpose foundation models on low-level vision. Q-Bench/Q-Bench+是一个面向多模态大模型底层视觉理解的数据集。此数据集从底层视觉的感知、描述、评价能力出发来对多模态大模型进行完整的测试，测试的对象既包括单张图像也包括图像对。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1000", "languages": [], "modality": null, "name": "Q-Bench", "openness": "unknown", "publisher": null, "released": "2024-08-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1000-q-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Q-Bench", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "QASC is a question-answering dataset with a focus on sentence composition. It consists of 9,980 8-way multiple-choice questions about grade school science (8,134 train, 926 dev, 920 test), and comes with a corpus of 17M sentences. QASC 是一个专注于句子组合的问答数据集。它包含 9,980 道小学科学的多项选择题（8,134 道用于训练，926 道用于开发，920 道用于测试），并配有一个包含 1,700 万个句子的语料库。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1103", "languages": [], "modality": null, "name": "QASC", "openness": "open", "publisher": "Allen Institute for AI", "released": "2020-02-04", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1103-qasc", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/QASC", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QASPER is a dataset of 5,049 information-seeking questions and answers anchored in 1,585 NLP research papers. Questions are written by NLP practitioners who read only titles and abstracts, while answers require understanding the full paper text and provide supporting evidence. The dataset challenges models with complex reasoning across document sections for academic document question answering. Each question seeks information present in the full text and is answered by a separate set of NLP practitioners who also provide supporting evidence to answers.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qasper:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qasper", "languages": [], "modality": "text", "name": "Qasper", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 41.9, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.419, "raw_min": 0.4, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qasper:phi-3.5-mini-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500"}, "unit": null}, "slug": "llm-stats-qasper", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "summarization"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QMSum is a benchmark for query-based multi-domain meeting summarization consisting of 1,808 query-summary pairs over 232 meetings across academic, product, and committee domains. The dataset enables models to select and summarize relevant spans of meetings in response to specific queries. Published at NAACL 2021, QMSum presents significant challenges in long meeting summarization where models must identify and summarize relevant content based on user queries.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qmsum:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qmsum", "languages": [], "modality": "text", "name": "QMSum", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 21.3, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.213, "raw_min": 0.199, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qmsum:phi-3.5-mini-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500"}, "unit": null}, "slug": "llm-stats-qmsum", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QVHighlights is a video moment retrieval benchmark for detecting moments and highlights in videos via natural language queries. Given a query, the model must localize the start and end times of relevant moments in the video, evaluated using metrics such as Recall@1 at a 0.5 IoU threshold.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qvhighlights:nova-2-lite", "reported_at": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qvhighlights", "languages": [], "modality": "multimodal", "name": "QVHighlights", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.772, "raw_min": 0.767, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qvhighlights:nova-2-lite", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500"}, "unit": null}, "slug": "llm-stats-qvhighlights", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QwenClawBench is a real-user-distribution Claw agent benchmark for evaluating coding agents on realistic developer tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qwenclawbench:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qwenclawbench", "languages": [], "modality": "text", "name": "QwenClawBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.618, "raw_min": 0.618, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qwenclawbench:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500"}, "unit": null}, "slug": "llm-stats-qwenclawbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QwenQoderBench is Qwen's internal benchmark for evaluating coding-agent performance.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qwen-qoder-bench:qwen3.8-max", "reported_at": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qwen-qoder-bench", "languages": [], "modality": "text", "name": "QwenQoderBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 58.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.584, "raw_min": 0.584, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qwen-qoder-bench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-qwen-qoder-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QwenReactBench is Qwen's internal React application generation benchmark, reported as a BT/Elo rating.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qwen-react-bench:qwen3.8-max", "reported_at": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qwen-react-bench", "languages": [], "modality": "multimodal", "name": "QwenReactBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 1724.0, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 1724.0, "raw_min": 1724.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qwen-react-bench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-qwen-react-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QwenSVG is Qwen's internal SVG generation benchmark for evaluating front-end and visual code generation. Scores are reported as BT/Elo ratings from auto-rendered outputs judged by a multimodal evaluator.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qwen-svg:qwen3.7-max", "reported_at": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qwen-svg", "languages": [], "modality": "multimodal", "name": "QwenSVG", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 1713.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 1713.0, "raw_min": 1608.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qwen-svg:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500"}, "unit": null}, "slug": "llm-stats-qwen-svg", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QwenSWEBench is Qwen's software-engineering agent benchmark for repository-level issue resolution.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qwen-swe-bench:qwen3.8-max", "reported_at": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qwen-swe-bench", "languages": [], "modality": "text", "name": "QwenSWEBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.7, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.807, "raw_min": 0.79, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qwen-swe-bench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-qwen-swe-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QwenWebBench is an internal front-end code generation benchmark by Qwen. It is bilingual (EN/CN) and spans 7 categories (Web Design, Web Apps, Games, SVG, Data Visualization, Animation, and 3D), using auto-render plus a multimodal judge for code and visual correctness. Scores are reported as BT/Elo ratings.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qwenwebbench:qwen3.6-27b", "reported_at": "2026-04-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qwenwebbench", "languages": [], "modality": "multimodal", "name": "QwenWebBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 1568.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 1568.0, "raw_min": 1487.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qwenwebbench:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500"}, "unit": null}, "slug": "llm-stats-qwenwebbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "simulation", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "QwenWorldBench is Qwen's internal benchmark for evaluating LLMs as world models that simulate agentic environments across Terminal, SWE, MCP, Search, OS, Android, and Web domains.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:qwenworldbench:qwen3.7-max", "reported_at": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:qwenworldbench", "languages": [], "modality": "text", "name": "QwenWorldBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.1, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.621, "raw_min": 0.573, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:qwenworldbench:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500"}, "unit": null}, "slug": "llm-stats-qwenworldbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RACE is a large-scale reading comprehension dataset with more than 28,000 passages and nearly 100,000 questions. The dataset is collected from English examinations in China, which are designed for middle school and high school students. RACE 是一个大规模的阅读理解数据集，包含超过 28,000 个段落和近 100,000 个问题。该数据集是从中国的英语考试中收集而来，这些考试是为中学和高中学生设计的。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:516", "languages": ["English"], "modality": null, "name": "RACE(High)", "openness": "unknown", "publisher": null, "released": "2017-04-15", "released_reference": {"basis": "paper_first_version", "note": "The RACE introduction defines its high-school and middle-school partitions together.", "source_key": "opencompass:516", "source_url": "https://arxiv.org/abs/1704.04683"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-516-race-high", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28High%29", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RACE is a large-scale reading comprehension dataset with more than 28,000 passages and nearly 100,000 questions. The dataset is collected from English examinations in China, which are designed for middle school and high school students. RACE 是一个大规模的阅读理解数据集，包含超过 28,000 个段落和近 100,000 个问题。该数据集是从中国的英语考试中收集而来，这些考试是为中学和高中学生设计的。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:517", "languages": ["English"], "modality": null, "name": "RACE(Middle)", "openness": "unknown", "publisher": null, "released": "2017-04-15", "released_reference": {"basis": "paper_first_version", "note": "The RACE introduction defines its middle-school and high-school partitions together.", "source_key": "opencompass:517", "source_url": "https://arxiv.org/abs/1704.04683"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-517-race-middle", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28Middle%29", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "科学智能", "AI for Science", "代码工程", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RDB2G-Bench provides comprehensive performance evaluation data for graph neural network models applied to relational database tasks. The dataset contains extensive experiments across multiple graph configurations and architectures. RDB2G-Bench提供了针对关系数据库任务应用的图神经网络模型的全面性能评估数据。该数据集包含了跨多种图配置和架构的广泛实验。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1918", "languages": [], "modality": null, "name": "RDB2G-Bench", "openness": "open", "publisher": "Kumo.AI , Kim Jaechul Graduate School of AI, KAIST", "released": "2025-06-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1918-rdb2g-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RDB2G-Bench", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "代码", "Code", "智能体", "Agent", "任务执行", "Task Execution", "逻辑推理", "Reasoning", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RE-Bench (Research Engineering Benchmark, v1) consists of 7 challenging, open-ended ML research engineering environments and data from 71 8-hour attempts by 61 distinct human experts. RE-Bench用于评估AI智能体研发的自动化能力，它由61位人类专家71次在7个具有挑战性的开放式ML研究工程环境中的8小时尝试的数据组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1252", "languages": [], "modality": null, "name": "RE-Bench", "openness": "unknown", "publisher": "Model Evaluation and Threat Research", "released": "2024-11-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1252-re-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RE-Bench", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Benchmarking Autonomous Agents on Deterministic Simulations of Real Websites 在真实网站的确定性模拟上对自主代理进行基准测试", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1755", "languages": [], "modality": null, "name": "REAL", "openness": "unknown", "publisher": "The AGI Company, Stanford University，etc.", "released": "2025-04-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1755-real", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/REAL", "unit": null}, {"aliases": [], "categories": ["multimodal", "document_understanding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RealKIE-FCC is a key information extraction benchmark drawn from real enterprise documents (FCC filings), part of the RealKIE suite of five novel datasets for enterprise key information extraction. Models must convert documents to markdown and extract structured fields against a specified JSON schema. Nova 2 reports results on a human-verified version of the dataset.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:realkie-fcc:nova-2-lite", "reported_at": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:realkie-fcc", "languages": [], "modality": "multimodal", "name": "RealKIE-FCC", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.67, "raw_min": 0.598, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:realkie-fcc:nova-2-pro", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500"}, "unit": null}, "slug": "llm-stats-realkie-fcc", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RealToxicityPrompts is a dataset of 100K naturally occurring, sentence-level prompts derived from a large corpus of English web text, paired with toxicity scores from a widely used toxicity classiﬁer. RealToxicityPrompts 是一个包含 100,000 个自然出现的、句子级提示的数据集，这些提示来自于大量的英语网络文本，并配有来自广泛使用的毒性分类器的毒性评分。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1124", "languages": ["English"], "modality": null, "name": "RealToxicityPrompts", "openness": "open", "publisher": "Allen Institute for Artiﬁcial Intelligence", "released": "2020-11-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1124-realtoxicityprompts", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RealToxicityPrompts", "unit": null}, {"aliases": [], "categories": ["spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RealWorldQA is a benchmark designed to evaluate basic real-world spatial understanding capabilities of multimodal models. The initial release consists of over 700 anonymized images taken from vehicles and other real-world scenarios, each accompanied by a question and easily verifiable answer. Released by xAI as part of their Grok-1.5 Vision preview to test models' ability to understand natural scenes and spatial relationships in everyday visual contexts.", "evidence_summary": {"document_count": 1, "model_count": 29, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:realworldqa:grok-1.5v", "reported_at": "2024-04-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": true, "key": "llm-stats:realworldqa", "languages": [], "modality": "multimodal", "name": "RealWorldQA", "openness": "unknown", "publisher": "XAI", "released": "2024-04-12", "released_reference": {"basis": "release_announcement", "note": "The dated Grok-1.5V announcement introduces RealWorldQA and provides its initial public dataset download.", "source_key": "opencompass:1361", "source_url": "https://x.ai/news/grok-1.5v"}, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 29, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 29, "model_count_basis": "source_model_id", "numeric_count": 29, "raw_max": 0.88, "raw_min": 0.622, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:realworldqa:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500"}, "unit": null}, "slug": "llm-stats-realworldqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RealWorldQA is a benchmark designed for real-world understanding, including 765 images, each accompanied by a question and a verifiable answer. RealWorldQA用于评估多模态模型在现实世界中的空间理解能力，包含765张图像，每张图像都配有一个问题和易于验证的答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": true, "key": "opencompass:1361", "languages": [], "modality": "multimodal", "name": "RealworldQA", "openness": "unknown", "publisher": "XAI", "released": "2024-04-12", "released_reference": {"basis": "release_announcement", "note": "The dated Grok-1.5V announcement introduces RealWorldQA and provides its initial public dataset download.", "source_key": "opencompass:1361", "source_url": "https://x.ai/news/grok-1.5v"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1361-realworldqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RealworldQA", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ReCoRD is a reading comprehension task, which requires to extract the answer from the article based on the given news article and question. ReCoRD是一个阅读理解任务，要求根据给定的新闻文章和问题，从文章中抽取出答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:530", "languages": [], "modality": null, "name": "ReCoRD", "openness": "unknown", "publisher": null, "released": "2018-10-30", "released_reference": {"basis": "paper_first_version", "note": "The original ReCoRD introduction, not the later SuperGLUE paper that reuses it.", "source_key": "opencompass:530", "source_url": "https://arxiv.org/abs/1810.12885"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-530-record", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ReCoRD", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RecreationBench is Qwen's long-horizon application-recreation benchmark for hybrid agents across Ubuntu, macOS, Windows, Android, and the web.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:recreationbench:qwen3.8-27b", "reported_at": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:recreationbench", "languages": [], "modality": "multimodal", "name": "RecreationBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 47.099999999999994, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.471, "raw_min": 0.471, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:recreationbench:qwen3.8-27b", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500"}, "unit": null}, "slug": "llm-stats-recreationbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "智能体", "Agent", "任务执行", "Task Execution", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RedCode provides comprehensive and practical evaluations on the safety of code agents, including 4,050 risky test cases covering 25 types of critical vulnerabilities spanning 8 domains and 160 prompts aiming to generate harmful code or software. RedCode旨在为LLM代码智能体的安全性提供全面实用的评估，包括来自8个领域25种关键漏洞的4050个风险测试用例，以及160个生成有害代码的提示。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1333", "languages": [], "modality": null, "name": "RedCode", "openness": "unknown", "publisher": "University of Chicago", "released": "2024-11-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1333-redcode", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RedCode", "unit": null}, {"aliases": [], "categories": ["spatial_reasoning", "grounding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RefCOCO-avg measures object grounding accuracy averaged across RefCOCO, RefCOCO+, and RefCOCOg benchmarks.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:refcoco-avg:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:refcoco-avg", "languages": [], "modality": "image", "name": "RefCOCO-avg", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.935, "display_multiplier": 1, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.935, "raw_min": 0.732, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:refcoco-avg:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500"}, "unit": null}, "slug": "llm-stats-refcoco-avg", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "grounding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RefCOCOg is a referring expression comprehension benchmark that evaluates spatial grounding in images. Given a natural language expression describing an object, the model must localize the correct region, evaluated by accuracy at a 0.5 IoU threshold. It features longer, more descriptive expressions than RefCOCO and RefCOCO+.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:refcocog:nova-2-omni", "reported_at": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:refcocog", "languages": [], "modality": "multimodal", "name": "RefCOCOg", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.863, "raw_min": 0.863, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:refcocog:nova-2-omni", "reported_date": "2025-12-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500"}, "unit": null}, "slug": "llm-stats-refcocog", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500", "unit": null}, {"aliases": [], "categories": ["spatial_reasoning", "grounding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RefSpatialBench evaluates spatial reference understanding and grounding.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:refspatialbench:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:refspatialbench", "languages": [], "modality": "image", "name": "RefSpatialBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.7, "display_multiplier": 1, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.7, "raw_min": 0.635, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:refspatialbench:qwen3.6-27b", "reported_date": "2026-04-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500"}, "unit": null}, "slug": "llm-stats-refspatialbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "大语言模型", "LLM", "知识储备", "Knowledge", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RepLiQA is suited for question-answering and topic retrieval tasks, including collection of five splits of test sets. Accurate answers can only be generated if a model can find relevant content within the provided document. RepLiQA适用于问答和主题检索任务，集合了是5个测试集；只有当模型可以在提供的文档中找到相关内容时，才能生成准确的答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1317", "languages": [], "modality": null, "name": "RepLiQA", "openness": "open", "publisher": "ServiceNow Research", "released": "2024-06-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1317-repliqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RepLiQA", "unit": null}, {"aliases": [], "categories": ["agents", "coding"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Repo Env evaluates an agent's ability to set up, configure, and run real repositories, including dependency resolution and environment management.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:repo-env:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:repo-env", "languages": [], "modality": "text", "name": "Repo Env", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 55.00000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.55, "raw_min": 0.467, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:repo-env:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500"}, "unit": null}, "slug": "llm-stats-repo-env", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RepoBench is a benchmark for evaluating repository-level code auto-completion systems through three interconnected tasks: RepoBench-R (retrieval of relevant code snippets across files), RepoBench-C (code completion with cross-file and in-file context), and RepoBench-P (pipeline combining retrieval and prediction). Supports Python and Java programming languages and addresses the gap in evaluating real-world, multi-file programming scenarios by providing a more complete comparison of performance in auto-completion systems.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:repobench:codestral-22b", "reported_at": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:repobench", "languages": [], "modality": "text", "name": "RepoBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 34.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.34, "raw_min": 0.34, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:repobench:codestral-22b", "reported_date": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500"}, "unit": null}, "slug": "llm-stats-repobench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RepoQA is a benchmark for evaluating long-context code understanding capabilities of Large Language Models through the Searching Needle Function (SNF) task, where LLMs must locate specific functions in code repositories using natural language descriptions. The benchmark contains 500 code search tasks spanning 50 repositories across 5 modern programming languages (Python, Java, TypeScript, C++, and Rust), tested on 26 general and code-specific LLMs to assess their ability to comprehend and navigate code repositories.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:repoqa:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:repoqa", "languages": [], "modality": "text", "name": "RepoQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.85, "raw_min": 0.77, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:repoqa:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500"}, "unit": null}, "slug": "llm-stats-repoqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["research", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ResearchClawBench evaluates research agents on realistic, tool-using research tasks that require code execution and filesystem workspace interaction.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:researchclawbench:mimo-v2.5", "reported_at": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:researchclawbench", "languages": [], "modality": "text", "name": "ResearchClawBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 16.91, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.1691, "raw_min": 0.1691, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:researchclawbench:mimo-v2.5", "reported_date": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500"}, "unit": null}, "slug": "llm-stats-researchclawbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "REVEAL(ReasoningVerification Evaluation) is a new dataset to benchmark automatic verifiers of complex Chain-ofThought reasoning in open-domain question answering settings. Reveal 是一个用于基准测试开放域问答环境中复杂链式推理自动验证器的新数据集。Reveal 包含关于语言模型答案中每个推理步骤的相关性、证据段落的归因和逻辑正确性的全面标签，涵盖多种数据集和最先进的语言模型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1087", "languages": [], "modality": null, "name": "Reveal", "openness": "restricted", "publisher": "Google", "released": "2024-05-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1087-reveal", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Reveal", "unit": null}, {"aliases": [], "categories": ["指令跟随", "Instruct", "其他", "Other", "大语言模型", "LLM", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RewardBench is a benchmark designed to evaluate the capabilities and safety of reward models (including those trained with Direct Preference Optimization, DPO). RewardBench是一个用于评估奖励模型（包括通过直接偏好优化（DPO）训练的模型）能力和安全性的基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1915", "languages": [], "modality": null, "name": "RewardBench", "openness": "restricted", "publisher": "Allen Institute for Artificial Intelligence , University of Washington , Cohere", "released": "2025-06-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1915-rewardbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RewardBench", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RExBench is a benchmark for evaluating large language model agents’ ability to autonomously implement AI research extensions, focusing on code generation, experimental design, and research comprehension. RExBench 是一个评估大型语言模型代理在自动实现 AI 研究扩展能力的基准，涵盖代码生成、实验设计和研究理解等维度。该基准包含12个基于真实论文和代码库的任务，每项任务由领域专家提供扩展指令，并通过自动化基础设施执行代理输出以验证成功标准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2033", "languages": [], "modality": null, "name": "RExBench", "openness": "unknown", "publisher": "University College London , Boston University , University of Vienna", "released": "2025-06-27", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2033-rexbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RExBench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "综合能力", "Comprehensive Capability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RFUAV offers a comprehensive benchmark dataset for Radio-Frequency (RF)-based drone detection and identification. RFUAV 提供了一个基于射频（RF）的无人机检测和识别的全面基准数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1656", "languages": [], "modality": null, "name": "RFUAV", "openness": "open", "publisher": "ZSTU, etc.", "released": "2025-03-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1656-rfuav", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RFUAV", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "长文本", "Long-Context", "任务执行", "Task Execution", "长上下文", "Long Context", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Artificial intelligence is undergoing the paradigm shift from closed language models to interconnected agent systems capable of external perception and information integration. As a representative embodiment, Deep Research Agents (DRAs) systematically exhibit the capabilities for task decomposition, Artificial intelligence is undergoing the paradigm shift from closed language models to interconnected agent systems capable of external perception and information integration. As a representative embodiment, Deep Research Agents (DRAs) systematically exhibit the capabilities for task decomposition,", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2371", "languages": [], "modality": null, "name": "RigorousBench", "openness": "unknown", "publisher": "Shanghai Artificial Intelligence Laboratory", "released": "2025-10-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2371-rigorousbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RigorousBench", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "指令跟随", "Instruct", "多模态模型", "VLM", "视觉生成", "Visual Generation", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RISEBench is a benchmark for evaluating large multimodal models (LMMs) on reasoning-informed visual editing tasks, targeting models with image understanding and generation capabilities. RISEBench 是一个用于评估多模态大模型（LMMs）在推理驱动视觉编辑任务中能力的基准，面向具备图像理解与生成能力的模型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2048", "languages": [], "modality": null, "name": "RISEBench", "openness": "restricted", "publisher": "Shanghai Jiao Tong University,Shanghai AI Laboratory,Wuhan University,etc.", "released": "2025-04-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2048-risebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RISEBench", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "VQA", "大语言模型", "LLM", "语言生成", "Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RM-Bench, a benchmark dataset for evaluating reward models of language modeling. RM-Bench，一个用于评估语言模型奖励模型的基准数据集", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1524", "languages": [], "modality": null, "name": "RM-Bench", "openness": "restricted", "publisher": "FDU, THU, Hong Kong University of Science and Technology", "released": "2024-10-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1524-rm-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RM-Bench", "unit": null}, {"aliases": [], "categories": ["robotics", "spatial_reasoning", "embodied", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RoboSpatialHome evaluates spatial understanding for robotic home navigation and manipulation.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:robospatialhome:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:robospatialhome", "languages": [], "modality": "image", "name": "RoboSpatialHome", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.739, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.739, "raw_min": 0.739, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:robospatialhome:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500"}, "unit": null}, "slug": "llm-stats-robospatialhome", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "instruction_following"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Robust IF evaluates instruction-following robustness on diverse, hard prompts, measuring whether a model reliably adheres to constraints across challenging single-turn and multi-turn scenarios.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:robust-if:mai-code-1-flash", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:robust-if", "languages": [], "modality": "text", "name": "Robust IF", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.199999999999996, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.612, "raw_min": 0.612, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:robust-if:mai-code-1-flash", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500"}, "unit": null}, "slug": "llm-stats-robust-if", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "ACL 2024", "大语言模型", "LLM", "语言生成", "Generation", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RoleLLM is a role-playing framework of data construction and evaluation (RoleBench), as well as solutions for both closed-source and open-source models (RoleGPT, RoleLLaMA, RoleGLM). We also propose Context-Instruct for long-text knowledge extraction and role-specific knowledge injection. RoleLLM 是一个角色扮演的数据构建和评估框架，同时提供闭源和开源模型的解决方案（RoleGPT、RoleLLaMA、RoleGLM）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1091", "languages": [], "modality": null, "name": "RoleLLM", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1091-rolellm", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RoleLLM", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code", "systems"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The RSI (Recursive Self-Improvement) Index is OpenAI's aggregate metric across a bundle of internal AI-research evaluations, including debugging research systems, optimizing kernels and training recipes, and improving other models, measuring progress toward recursive self-improvement.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:rsi-index:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:rsi-index", "languages": [], "modality": "text", "name": "RSI Index", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.9, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.579, "raw_min": 0.419, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:rsi-index:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500"}, "unit": null}, "slug": "llm-stats-rsi-index", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500", "unit": null}, {"aliases": ["RSI-Bench", "RSI Bench", "RSIBench", "Recursive Self-Improvement Benchmark"], "categories": ["agent"], "collected_at": null, "description": "Announced by Scale AI on 2026-08-07 and still in preview: it is soliciting task contributions rather than reporting results, so no model card in this registry reports it and its adoption count is zero. That zero is a real reading, not a gap in the crawl. Recorded now so the benchmark is addressable by name before any score exists, and so the first card to report it has an id to resolve against. Scores are per-task against a baseline under a fixed compute budget, graded on reliability, efficiency, generality and idea quality, so a single headline number should not be expected. Not to be confused with the llm-stats \"RSI Index\" record, which is a different instrument.", "evidence_summary": {"document_count": null, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:rsi_bench", "languages": [], "modality": null, "name": "RSI-Bench", "openness": "unknown", "publisher": null, "released": "2026-08-07", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:rsi_bench", "source_url": "https://www.rsi-benchmark.com/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "rsi_bench", "source": "model_reports", "source_url": "https://www.rsi-benchmark.com/", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "This dataset follows a similar procedure to the original MMVP benchmark on natural images but directed towards the remote sensing domain. Challenging visual patterns are identified based on CLIP blind pairs, accompanied with the correpsonding questions, options and ground-truth answer. RSMMVP遵循与原始 MMVP 基准在自然图像上的类似流程，但针对遥感领域。根据 CLIP 盲对识别具有挑战性的视觉模式，并附带相应的问题、选项和真实答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1678", "languages": [], "modality": "multimodal", "name": "RSMMVP", "openness": "restricted", "publisher": "aimasammi, etc.", "released": "2025-03-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1678-rsmmvp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RSMMVP", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RTE is a natural language inference task, which requires to determine the logical relation between the given sentence pair, with three relations: entailment, contradiction and neutral. RTE是一个自然语言推理任务，要求根据给定的句子对，判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": "2019-05-02", "first_score_source_reference": {"basis": "score_publication", "date_precision": "score_publication", "note": "The original benchmark release day is absent from the crawl. Table 2 provides the earliest dated numeric language-model evaluation retained in this date archive for the source-linked RTE task; it is not a benchmark release date.", "reported_at": "2019-05-02", "score_evidence": {"locator": "Table 2, BERT row, RTE column", "metric": "accuracy", "model": "BERT-large-cased (fine-tuned)", "unit": "percent", "value": 70.1}, "source_key": "opencompass:528", "source_url": "https://arxiv.org/abs/1905.00537v1"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:528", "languages": [], "modality": null, "name": "RTE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-528-rte", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RTE", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RULER v1 is a synthetic long-context benchmark for measuring how model quality degrades as input length increases. This packaging follows the public standalone NVIDIA RULER implementation with 13 official tasks spanning retrieval, multi-hop tracing, aggregation, and QA.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ruler:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ruler", "languages": [], "modality": "text", "name": "RULER", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.69999999999999, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.947, "raw_min": 0.841, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ruler:nemotron-3-ultra-550b-a55b", "reported_date": "2026-06-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500"}, "unit": null}, "slug": "llm-stats-ruler", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RULER 1000K evaluates the official 13-task RULER v1 suite at a 1048576-token (1M) context budget.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ruler-1000k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ruler-1000k", "languages": [], "modality": "text", "name": "RULER 1000K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.863, "raw_min": 0.863, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ruler-1000k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500"}, "unit": null}, "slug": "llm-stats-ruler-1000k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RULER 128k evaluates the official 13-task RULER v1 suite at a 131072-token context budget.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ruler-128k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ruler-128k", "languages": [], "modality": "text", "name": "RULER 128k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.37, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.8937, "raw_min": 0.8937, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ruler-128k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500"}, "unit": null}, "slug": "llm-stats-ruler-128k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RULER 2048K evaluates the official 13-task RULER v1 suite at a 2097152-token (2M) context budget.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ruler-2048k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ruler-2048k", "languages": [], "modality": "text", "name": "RULER 2048K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.816, "raw_min": 0.816, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ruler-2048k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500"}, "unit": null}, "slug": "llm-stats-ruler-2048k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RULER 512K evaluates the official 13-task RULER v1 suite at a 524288-token context budget.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ruler-512k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ruler-512k", "languages": [], "modality": "text", "name": "RULER 512K", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.871, "raw_min": 0.871, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ruler-512k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500"}, "unit": null}, "slug": "llm-stats-ruler-512k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "RULER 64k evaluates the official 13-task RULER v1 suite at a 65536-token context budget.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:ruler-64k:minicpm-sala", "reported_at": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:ruler-64k", "languages": [], "modality": "text", "name": "RULER 64k", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 92.65, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.9265, "raw_min": 0.9265, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:ruler-64k:minicpm-sala", "reported_date": "2026-02-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500"}, "unit": null}, "slug": "llm-stats-ruler-64k", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "Multimodal", "多模态模型", "VLM", "音频理解", "Audio Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RUListening: Robust Understanding through Listening, an automated QA generation framework for evaluating multimodal perception. RUListening：通过聆听的稳健理解，这是一个用于评估多模态感知的自动问答生成框架。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1738", "languages": [], "modality": null, "name": "RUListening", "openness": "unknown", "publisher": "Independent Researcher, University of California, San Diego", "released": "2025-04-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1738-rulistening", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RUListening", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "科学智能", "AI for Science", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "RxRx3-core dataset is a challenge dataset in phenomics optimized for the research\ncommunity. RxRx3-core数据集是Recursion为研究社区优化的表型组学挑战数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1707", "languages": [], "modality": null, "name": "RXRX3-CORE", "openness": "unknown", "publisher": "Universitat de les Illes Balears, etc.", "released": "2025-03-26", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1707-rxrx3-core", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RXRX3-CORE", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "LLM", "大语言模型", "安全对齐", "Safety Alignment", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "S-Eval is a new comprehensive, multi-dimensional and open-ended safety evaluation benchmark for LLMs consisting of 220,000 evaluation prompts (still in active expansion) across 102 risk subcategories and 10 advanced jailbreak attacks. S-Eval 是一个针对 LLM 的全新全面、多维、开放式安全评估基准，包含 102 个风险子类别的 220,000 个评估提示（仍在积极扩展中）和 10 个高级越狱攻击。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1003", "languages": ["English", "Chinese"], "modality": null, "name": "S-Eval", "openness": "restricted", "publisher": "Zhejiang University & Alibaba", "released": "2024-05-23", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1003-s-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/S-Eval", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "理解", "Understanding", "指令跟随", "Instruct", "系统1", "快思考", "LRM", "大语言模型", "LLM", "逻辑推理", "Reasoning", "语言理解", "Comprehension", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "S1-Bench is a novel benchmark designed to evaluate the performance of LRMs on simple tasks that are more aligned with intuitive System 1 thinking, rather than deliberate System 2 reasoning. S1-Bench offers a set of simple, diverse, and naturally clear questions across multiple domains and languages. S1-Bench是一个新颖的基准，旨在评估大模型在简单任务中的表现，这些任务更倾向于直观的系统1思维，而非深思熟虑的系统2推理。尽管大模型在复杂推理任务中通过明确的思维链取得了显著突破，但它们对深度分析思维的依赖可能限制了其系统1思维能力。此外，目前缺乏评估大模型在需要此类能力的任务中表现的基准。为了填补这一空白，S1-Bench提供了一组简单、多样且自然清晰的问题，涵盖多个领域和语言，专门设计用于评估大模型在此类任务中的表现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1772", "languages": ["English", "Chinese"], "modality": null, "name": "S1-Bench", "openness": "open", "publisher": "中国科学院", "released": "2025-04-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1772-s1-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/S1-Bench", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SAEBench is a comprehensive benchmark designed to evaluate and compare the performance of language model sparse autoencoders (SAEs). This benchmark provides over 200 SAE models covering seven architectures for systematic comparison, and has open-source code and models. SAEBench 是一个旨在评估和比较语言模型稀疏自动编码器（SAEs）性能的综合性基准。该基准提供超过200个SAE模型，涵盖七种架构，以实现系统性比较，并且开源了代码和模型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2083", "languages": [], "modality": null, "name": "SAEBench", "openness": "open", "publisher": "Independent , Decode Research , University College London , etc.", "released": "2025-03-12", "released_reference": {"basis": "paper_first_version", "note": "SAEBench's own introduction, not the papers introducing its component evaluation methods.", "source_key": "opencompass:2083", "source_url": "https://arxiv.org/abs/2503.09532"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2083-saebench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SAEBench", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "ACL 2024", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SafetyBench, a comprehensive benchmark for evaluating the safety of LLMs, which comprises 11,435 diverse multiple choice questions spanning across 7 distinct categories of safety concerns. Notably, SafetyBench also incorporates both Chinese and English data. SafetyBench 是一个全面的基准，用于评估大型语言模型（LLMs）的安全性，包含 11,435 道多样化的选择题，涵盖 7 个不同的安全关注类别。SafetyBench 还包含中文和英文的数据，方便双语评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1073", "languages": ["English", "Chinese"], "modality": null, "name": "SafetyBench", "openness": "unknown", "publisher": "thu-coai", "released": "2024-06-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1073-safetybench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SafetyBench", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "ACL 2024", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SALAD-Bench, a safety benchmark specifically designed for evaluating LLMs, attack, and defense methods. Distinguished by its breadth, SALAD-Bench transcends conventional benchmarks through its large scale, rich diversity, intricate taxonomy\nspanning three levels, and versatile functionalities. SALAD-Bench 是一个专门用于评估大型语言模型（LLMs）、攻击和防御方法的安全基准。SALAD-Bench 的特点在于其广泛性，超越了传统基准，具有大规模、丰富的多样性、复杂的三层分类法以及多功能性。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1077", "languages": [], "modality": null, "name": "SALAD-Bench", "openness": "unknown", "publisher": "Shanghai AI Laboratory", "released": "2024-02-07", "released_reference": {"basis": "paper_first_version", "note": "First version introducing SALAD-Bench and its hierarchical safety evaluation.", "source_key": "opencompass:1077", "source_url": "https://arxiv.org/abs/2402.05044"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1077-salad-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SALAD-Bench", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SAT Math benchmark from AGIEval containing standardized mathematics questions from the College Board SAT examination, designed to evaluate mathematical reasoning capabilities of foundation models using human-centric assessment methods.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:sat-math:gpt-4-0613", "reported_at": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:sat-math", "languages": [], "modality": "text", "name": "SAT Math", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.89, "raw_min": 0.89, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:sat-math:gpt-4-0613", "reported_date": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500"}, "unit": null}, "slug": "llm-stats-sat-math", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "安全对齐", "Safety Alignment", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SCAM is the largest and most diverse dataset of real-world typographic attack images to date, containing 1,162 images across hundreds of object categories and attack words. SCAM，是迄今为止规模最大、多样性最丰富的真实世界排版攻击图像数据集，包含数百个对象类别和攻击词汇的1,162张图像。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1732", "languages": [], "modality": "multimodal", "name": "SCAM", "openness": "restricted", "publisher": "BLISS e.V. , Berliner Hochschule für Technik (BHT), etc.", "released": "2025-04-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1732-scam", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SCAM", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "coding"], "collected_at": "2026-08-25T10:41:06Z", "description": "Coding", "evidence_summary": {"document_count": 1, "model_count": 577, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:scicode:ddc748d0-6a9b-466b-8d6c-68417980d56d", "reported_at": "2023-07-11", "source_url": "https://artificialanalysis.ai/evaluations/scicode"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:scicode", "languages": [], "modality": null, "name": "SciCode", "openness": "unknown", "publisher": null, "released": "2024-07-18", "released_reference": {"basis": "paper_first_version", "note": "First version of the SciCode paper. The source evaluates SciCode subproblems with scientific background; it does not refer to the later SciCode-Verified benchmark.", "source_key": "artificial-analysis:scicode", "source_url": "https://arxiv.org/abs/2407.13168"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 577, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.1851851851852, "display_multiplier": 100, "model_count": 577, "model_count_basis": "source_model_id", "numeric_count": 577, "raw_max": 0.601851851851852, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:scicode:cd55210d-358e-4df1-ba9c-9acb5f186cc9", "reported_date": "2026-06-09", "source_url": "https://artificialanalysis.ai/evaluations/scicode"}, "unit": null}, "slug": "artificial-analysis-scicode", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/scicode", "unit": null}, {"aliases": [], "categories": ["math", "physics", "reasoning", "biology", "chemistry", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SciCode is a research coding benchmark curated by scientists that challenges language models to code solutions for scientific problems. It contains 338 subproblems decomposed from 80 challenging main problems across 16 natural science sub-fields including mathematics, physics, chemistry, biology, and materials science. Problems require knowledge recall, reasoning, and code synthesis skills.", "evidence_summary": {"document_count": 1, "model_count": 21, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:scicode:glm-4.5", "reported_at": "2025-07-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:scicode", "languages": [], "modality": "text", "name": "SciCode", "openness": "unknown", "publisher": null, "released": "2024-07-18", "released_reference": {"basis": "paper_first_version", "note": "The reviewed source identity explicitly links this introducing SciCode paper. No score or adoption measurement is transferred from another source.", "source_key": "llm-stats:scicode", "source_url": "https://arxiv.org/abs/2407.13168"}, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 21, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.8, "display_multiplier": 100, "model_count": 21, "model_count_basis": "source_model_id", "numeric_count": 21, "raw_max": 0.598, "raw_min": 0.326, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:scicode:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500"}, "unit": null}, "slug": "llm-stats-scicode", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500", "unit": null}, {"aliases": ["SciCode"], "categories": ["science"], "collected_at": null, "description": "Scientific code generation graded by test execution; subproblem context matters.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000scicode\u0000google_gemini_3_1_pro_model_card\u0000scicode\u0000Thinking (High)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "first_score_reported_at": "2026-02-19", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000scicode\u0000google_gemini_3_1_pro_model_card\u0000scicode\u0000Thinking (High)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:scicode", "languages": [], "modality": null, "name": "SciCode", "openness": "unknown", "publisher": null, "released": "2024-07-18", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:scicode", "source_url": "https://scicode-bench.github.io/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 59.0, "raw_min": 58.7, "source_reference": {"obs_id": "curated\u0000scicode\u0000google_gemini_3_1_pro_model_card\u0000scicode\u0000Thinking (High)\u0000Gemini 3.1 Pro", "observation_id": "curated\u0000scicode\u0000google_gemini_3_1_pro_model_card\u0000scicode\u0000Thinking (High)\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "reported_date": "2026-02-19", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "unit": "percent"}, "slug": "scicode", "source": "model_reports", "source_url": "https://scicode-bench.github.io/", "unit": "percent"}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ScienceQA is the first large-scale multimodal science question answering benchmark with 21,208 multiple-choice questions covering 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. The benchmark includes both text and image modalities, featuring detailed explanations and Chain-of-Thought reasoning to diagnose multi-hop reasoning ability.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:scienceqa:phi-3.5-vision-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:scienceqa", "languages": [], "modality": "multimodal", "name": "ScienceQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.913, "raw_min": 0.913, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:scienceqa:phi-3.5-vision-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500"}, "unit": null}, "slug": "llm-stats-scienceqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "科学智能", "AI for Science", "知识储备", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SCIENCEQA is a new benchmark that consists of ∼21k multimodal multiple choice questions with diverse science topics and annotations of their answers with corresponding lectures and explanations. SCIENCEQA 包含约 21,000 道多模态选择题，涵盖多种科学主题，并附有相应的讲座和解释的答案注释。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": false, "has_size": true, "key": "opencompass:1100", "languages": [], "modality": null, "name": "ScienceQA", "openness": "unknown", "publisher": "University of California, Los Angeles & Arizona State University & Allen Instit", "released": "2022-10-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1100-scienceqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ScienceQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ScienceQA Visual is a multimodal science question answering benchmark consisting of 21,208 multiple-choice questions from elementary and high school science curricula. The dataset covers 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. 48.7% of questions include image context requiring multimodal reasoning. Questions are annotated with lectures (83.9%) and explanations (90.5%) to support chain-of-thought reasoning for science question answering.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:scienceqa-visual:phi-4-multimodal-instruct", "reported_at": "2025-02-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:scienceqa-visual", "languages": [], "modality": "multimodal", "name": "ScienceQA Visual", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.975, "raw_min": 0.975, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:scienceqa-visual:phi-4-multimodal-instruct", "reported_date": "2025-02-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500"}, "unit": null}, "slug": "llm-stats-scienceqa-visual", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "知识", "Knowledge", "NeurIPS 2024", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SciFIBench is a scientific figure interpretation benchmark for LMMs, consisting of 2000 questions split between two tasks across 8 categories. SciFIBench用于评估多模态大模型的科学图表解释能力，由2000个问题组成，涵盖8个类别的2种任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1323", "languages": [], "modality": "multimodal", "name": "SciFIBench", "openness": "restricted", "publisher": "University of Cambridge", "released": "2024-05-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1323-scifibench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SciFIBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "spatial_reasoning", "grounding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ScreenSpot is the first realistic GUI grounding benchmark that encompasses mobile, desktop, and web environments. The dataset comprises over 1,200 instructions from iOS, Android, macOS, Windows and Web environments, along with annotated element types (text and icon/widget), designed to evaluate visual GUI agents' ability to accurately locate screen elements based on natural language instructions.", "evidence_summary": {"document_count": 1, "model_count": 16, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:screenspot:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:screenspot", "languages": [], "modality": "multimodal", "name": "ScreenSpot", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 16, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.8, "display_multiplier": 100, "model_count": 16, "model_count_basis": "source_model_id", "numeric_count": 16, "raw_max": 0.958, "raw_min": 0.833, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:screenspot:qwen3-vl-32b-instruct", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500"}, "unit": null}, "slug": "llm-stats-screenspot", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "spatial_reasoning", "grounding", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ScreenSpot-Pro is a novel GUI grounding benchmark designed to rigorously evaluate the grounding capabilities of multimodal large language models (MLLMs) in professional high-resolution computing environments. The benchmark comprises 1,581 instructions across 23 applications spanning 5 industries and 3 operating systems, featuring authentic high-resolution images from professional domains with expert annotations. Unlike previous benchmarks that focus on cropped screenshots in consumer applications, ScreenSpot-Pro addresses the complexity and diversity of real-world professional software scenarios, revealing significant performance gaps in current MLLM GUI perception capabilities.", "evidence_summary": {"document_count": 1, "model_count": 25, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:screenspot-pro:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:screenspot-pro", "languages": [], "modality": "multimodal", "name": "ScreenSpot Pro", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 25, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.9, "display_multiplier": 100, "model_count": 25, "model_count_basis": "source_model_id", "numeric_count": 25, "raw_max": 0.879, "raw_min": 0.29, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:screenspot-pro:claude-opus-4-8", "reported_date": "2026-05-28", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-screenspot-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "search"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Seal-0 is a benchmark for evaluating agentic search capabilities, testing models' ability to navigate and retrieve information using tools.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:seal-0:kimi-k2-thinking-0905", "reported_at": "2025-09-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:seal-0", "languages": [], "modality": "text", "name": "Seal-0", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.4, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.574, "raw_min": 0.414, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:seal-0:kimi-k2.5", "reported_date": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500"}, "unit": null}, "slug": "llm-stats-seal-0", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Search and Function-Calling is an OpenAI internal production benchmark measuring reliable search-tool use and function calling in agentic workflows, reported as a pass rate.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:openai-search-function-calling:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:openai-search-function-calling", "languages": [], "modality": "text", "name": "Search and Function-Calling", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.6, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.946, "raw_min": 0.897, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:openai-search-function-calling:gpt-5.6-terra", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500"}, "unit": null}, "slug": "llm-stats-openai-search-function-calling", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "智能体", "Agent", "任务执行", "Task Execution", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SEC-bench is a benchmark designed to evaluate large language model (LLM) agents on real-world software security tasks. SEC-bench 是一个面向大型语言模型（LLM）智能体的软件安全任务评测基准，旨在自动化评估模型在真实漏洞环境中的能力。 该基准通过多智能体框架自动构建代码仓库、复现漏洞并生成修复补丁，涵盖漏洞验证（PoC 生成）和补丁修复两个关键任务，包含数百个真实安全案例。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1986", "languages": [], "modality": null, "name": "SEC-bench", "openness": "open", "publisher": "University of Illinois Urbana-Champaign , Purdue University", "released": "2025-06-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1986-sec-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEC-bench", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SEC-bench Pro is a self-evolving software-security benchmark that measures agent bug hunting on critical, high-complexity systems. It instantiates validated vulnerabilities across the V8 and SpiderMonkey JavaScript engines as reproducible vulnerability-discovery and proof-of-concept-generation tasks with oracle-based validation.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:sec-bench-pro:gpt-5.6-luna", "reported_at": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:sec-bench-pro", "languages": [], "modality": "text", "name": "SEC-bench Pro", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.712, "raw_min": 0.489, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:sec-bench-pro:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-sec-bench-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Tencent Zhuque Lab and Tencent Security Keen Lab, together with Tencent Huyuan Team, Professor Jiang Yong/Professor Xia Shutao's team from Tsinghua University, Professor Luo Xiapu's research team from Hong Kong Polytechnic University, and OpenCompass team from Shanghai Artificial Intelligence Laboratory, have jointly built a safety benchmark, SecBench. We provide fair, impartial, objective, and comprehensive evaluation capabilities and promote the construction of large model in security dimension. 腾讯朱雀实验室和腾讯安全科恩实验室联合腾讯混元大模型团队、清华大学江勇教授/夏树涛教授团队、香港理工大学罗夏朴教授研究团队以及上海人工智能实验室OpenCompass团队，通过建设安全大模型评测基准SecBench，为安全大模型研发提供公平、公正、客观、全面的评测能力，推动安全大模型建设。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "opencompass:948", "languages": [], "modality": null, "name": "SecBench", "openness": "unknown", "publisher": null, "released": "2024-01-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-948-secbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SecBench", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SecCodeBench evaluates LLM coding agents on secure code generation and vulnerability detection, testing the ability to produce code that is both functional and free from security vulnerabilities.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:seccodebench:qwen3.5-397b-a17b", "reported_at": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:seccodebench", "languages": [], "modality": "text", "name": "SecCodeBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.30000000000001, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.683, "raw_min": 0.683, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:seccodebench:qwen3.5-397b-a17b", "reported_date": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500"}, "unit": null}, "slug": "llm-stats-seccodebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500", "unit": null}, {"aliases": ["SecCodeBench"], "categories": ["security"], "collected_at": null, "description": "Measures secure-coding behaviour, not exploitation capability.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000seccodebench\u0000qwen3_5_model_card\u0000seccodebench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000seccodebench\u0000qwen3_5_model_card\u0000seccodebench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:seccodebench", "languages": [], "modality": null, "name": "SecCodeBench", "openness": "unknown", "publisher": null, "released": "2025-10-14", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:seccodebench", "source_url": "https://github.com/alibaba/SecCodeBench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.3, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 68.3, "raw_min": 68.3, "source_reference": {"obs_id": "curated\u0000seccodebench\u0000qwen3_5_model_card\u0000seccodebench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000seccodebench\u0000qwen3_5_model_card\u0000seccodebench\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "seccodebench", "source": "model_reports", "source_url": "https://github.com/alibaba/SecCodeBench", "unit": "percent"}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SEED-Bench aims at the evaluation of generative comprehension in MLLMs, consisting of 19K multiple choice questions, which spans 12 evaluation dimensions including the comprehension of both the image and video modality. SEED-Bench用于评估多模态大模型的理解能力，包括对图像和视频的理解，由跨越12个评估维度的19K道多项选择题组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1359", "languages": [], "modality": "multimodal", "name": "SEED-Bench", "openness": "restricted", "publisher": "Tencent AI Lab", "released": "2023-07-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1359-seed-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SEED-Bench-2 assesses both text and image generation of MLLMs. It spans 27 evaluation dimensions, featuring 24K multiple-choice questions with precise human annotations. SEED-Bench-2用于评估多模态大模型的文本和图像生成能力，包括跨越27个维度的24K道多选题及准确的人工注释。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1363", "languages": [], "modality": "multimodal", "name": "SEED-Bench-2", "openness": "restricted", "publisher": "Tencent AI Lab", "released": "2023-11-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1363-seed-bench-2", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "text-rich image", "多模态模型", "VLM", "图像理解", "Image Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SEED-Bench-2-Plus, a benchmark specifically designed for evaluating text-rich visual comprehension of MLLMs. The benchmark comprises 2.3K multiple-choice questions with precise human annotations, spanning three broad categories: Charts, Maps, and Webs. SEED-Bench-2-Plus, a benchmark specifically designed for evaluating text-rich visual comprehension of MLLMs. The benchmark comprises 2.3K multiple-choice questions with precise human annotations, spanning three broad categories: Charts, Maps, and Webs.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1395", "languages": [], "modality": "multimodal", "name": "SEED-Bench-2-Plus", "openness": "restricted", "publisher": "Tencent AI Lab", "released": "2024-04-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1395-seed-bench-2-plus", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2-Plus", "unit": null}, {"aliases": [], "categories": ["agents", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SeedClawBench is an agentic coding benchmark measuring overall model performance on real-world, tool-using software development tasks.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:seedclawbench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:seedclawbench", "languages": [], "modality": "text", "name": "SeedClawBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.60000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.666, "raw_min": 0.638, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:seedclawbench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500"}, "unit": null}, "slug": "llm-stats-seedclawbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "VQA", "科学智能", "AI for Science", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The Scientists' First Exam (SFE) benchmark, designed to comprehensively evaluate the scientific cognitive capabilities of MLLMs through three cognitive levels (cog-levels):Scientific Signal Perception、Scientific Attribute Understanding 、Scientific Comparative Reasoning. The Scientists' First Exam (SFE) 基准测试旨在通过三个认知层级——科学信号感知、科学属性理解 和 科学对比推理，全面评估多模态大语言模型（MLLMs）的科学认知能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1961", "languages": ["English", "Chinese"], "modality": "multimodal", "name": "SFE", "openness": "unknown", "publisher": "Shanghai Artificial Intelligence Laboratory", "released": "2025-06-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1961-sfe", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SFE", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "NeurIPS 2024", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SG-Bench assess LLM safety across various tasks and prompt types. It integrates both generative and discriminative evaluation tasks and includes extended data to examine the impact of prompt engineering and jailbreak. SG-Bench用于评估LLM在不同任务和提示下的安全性，整合了生成性和判别性评估任务，并包含扩展数据以度量提示工程和越狱对安全性的影响。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1330", "languages": [], "modality": null, "name": "SG-Bench", "openness": "unknown", "publisher": "Peking University", "released": "2024-10-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1330-sg-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SG-Bench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "科学", "Science", "代码", "Code", "deep research", "reasoning", "lab protocol", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "代码工程", "逻辑推理", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SGI-Bench operationalizes this definition via four scientist-aligned task families: deep research, idea generation, dry/wet experiments, and multimodal experimental reasoning. The benchmark spans 10 disciplines and more than 1,000 expert-curated samples inspired by Science's 125 Big Questions. 科学通用智能（SGI）被定义为一种人工智能系统，其能够以接近人类科学家的通用性与熟练程度，自主地贯穿并迭代完整的科学研究流程，包括审思、构思、行动与感知四个阶段。SGI-Bench 通过四类与科学家工作流程对齐的任务对上述定义进行操作化刻画，具体包括：深度研究、思想与假设生成、干/湿实验、以及多模态实验推理。该评测基准覆盖10 个科学学科领域，包含1,000 余个由领域专家精心策划的样本，其设计灵感来源于 Science 杂志提出的 125 个重大科学问题（125 Big Questions）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2325", "languages": [], "modality": "multimodal", "name": "SGI-Bench", "openness": "unknown", "publisher": "上海人工智能实验室", "released": "2025-12-22", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2325-sgi-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SGI-Bench", "unit": null}, {"aliases": [], "categories": ["安全", "Safety", "Implicit Risk", "大语言模型", "LLM", "安全对齐", "Safety Alignment", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Targeting overlooked implicit risks in vertical domains (Education, Finance, Management), we introduce an implicit risk benchmark and the MENTOR framework. By leveraging Rule Evolution and Activation Steering, MENTOR effectively detects and mitigates these subtle hazards. Shell由华东师范大学Shell@Educhat团队和上海人工智能实验室联合推出。当下，确保垂直领域任务中大模型的安全性至关重要。虽然目前的对齐工作主要针对偏见和暴力等显性风险，但往往忽略了更深层次的特定领域隐性风险。研发团队推出了一个包含大量隐式风险查询的基准测试集，将风险分为引导、反思、禁止三类，以及 MENTOR 框架。该框架利用规则演化循环（REC）和激活引导（RV）技术，能够有效发现并缓解这些不易察觉的潜在风险。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2348", "languages": [], "modality": null, "name": "Shell", "openness": "unknown", "publisher": "华东师范大学 &上海人工智能实验室", "released": "2025-12-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2348-shell", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Shell", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "NeurIPS 2024", "大语言模型", "LLM", "知识储备", "Knowledge", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Shopping MMLU is a diverse multi-task online shopping benchmark derived from real-world Amazon data. It consists of 57 tasks covering 4 major shopping skills: concept understanding, knowledge reasoning, user behavior alignment, and multi-linguality. Shopping MMLU是一个基于真实亚马逊数据的多样化多任务在线购物基准测试，由57项任务组成，涵盖概念理解、知识推理、用户行为对齐和多语言4大购物场景技能。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1321", "languages": [], "modality": null, "name": "ShoppingMMLU", "openness": "unknown", "publisher": "Amazon.com", "released": "2024-10-28", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1321-shoppingmmlu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ShoppingMMLU", "unit": null}, {"aliases": [], "categories": ["structured_output", "instruction_following", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SIFO (Simple Instruction Following) evaluates how well language models follow simple, explicit instructions. It tests fundamental instruction-following capabilities across various task types.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:sifo:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:sifo", "languages": [], "modality": "text", "name": "SIFO", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.773, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.773, "raw_min": 0.773, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:sifo:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500"}, "unit": null}, "slug": "llm-stats-sifo", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500", "unit": null}, {"aliases": [], "categories": ["structured_output", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SIFO-Multiturn evaluates instruction following capabilities in multi-turn conversational settings, testing how well models maintain context and follow instructions across multiple exchanges.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:sifo-multiturn:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:sifo-multiturn", "languages": [], "modality": "text", "name": "SIFO-Multiturn", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.711, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.711, "raw_min": 0.711, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:sifo-multiturn:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500"}, "unit": null}, "slug": "llm-stats-sifo-multiturn", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "factuality", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SimpleQA is a factuality benchmark developed by OpenAI that measures the short-form factual accuracy of large language models. The benchmark contains 4,326 short, fact-seeking questions that are adversarially collected and designed to have single, indisputable answers. Questions cover diverse topics from science and technology to entertainment, and the benchmark also measures model calibration by evaluating whether models know what they know.", "evidence_summary": {"document_count": 1, "model_count": 47, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:simpleqa:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:simpleqa", "languages": [], "modality": "text", "name": "SimpleQA", "openness": "restricted", "publisher": "OpenAI", "released": "2024-10-30", "released_reference": null, "repo_kind": "harness_only", "repo_resolution_status": "resolved", "score_count": 47, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.1, "display_multiplier": 100, "model_count": 47, "model_count_basis": "source_model_id", "numeric_count": 47, "raw_max": 0.971, "raw_min": 0.018, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:simpleqa:deepseek-v3.2-exp", "reported_date": "2025-09-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500"}, "unit": null}, "slug": "llm-stats-simpleqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500", "unit": null}, {"aliases": ["SimpleQA", "SimpleQA Verified"], "categories": ["factuality"], "collected_at": null, "description": "Knowledge cutoff must be recorded alongside the score.", "evidence_summary": {"document_count": 9, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000simpleqa\u0000deepseek_v3_report\u0000simpleqa\u0000Correct\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000simpleqa\u0000deepseek_v3_report\u0000simpleqa\u0000Correct\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:simpleqa", "languages": [], "modality": null, "name": "SimpleQA", "openness": "unknown", "publisher": null, "released": "2024-10-30", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:simpleqa", "source_url": "https://openai.com/index/introducing-simpleqa/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 57.9, "display_multiplier": 1, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 57.9, "raw_min": 24.9, "source_reference": {"obs_id": "curated\u0000simpleqa\u0000deepseek_v4_model_card\u0000simpleqa_verified\u0000think max, pass@1\u0000DeepSeek-V4-Pro", "observation_id": "curated\u0000simpleqa\u0000deepseek_v4_model_card\u0000simpleqa_verified\u0000think max, pass@1\u0000DeepSeek-V4-Pro", "reported_at": "2026-04-22", "reported_date": "2026-04-22", "source_id": "deepseek_v4_model_card", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"}, "unit": "percent"}, "slug": "simpleqa", "source": "model_reports", "source_url": "https://openai.com/index/introducing-simpleqa/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "factuality", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SimpleQA Verified is a curated, reliability-focused subset of SimpleQA that addresses label noise and redundancy in the original benchmark, measuring short-form parametric factual accuracy of large language models on fact-seeking questions with single, indisputable answers.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:simpleqa-verified:mai-thinking-1", "reported_at": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:simpleqa-verified", "languages": [], "modality": "text", "name": "SimpleQA Verified", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 31.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.31, "raw_min": 0.206, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:simpleqa-verified:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500"}, "unit": null}, "slug": "llm-stats-simpleqa-verified", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "image_to_text", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SimpleVQA is a visual question answering benchmark focused on simple queries.", "evidence_summary": {"document_count": 1, "model_count": 14, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:simplevqa:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:simplevqa", "languages": [], "modality": "multimodal", "name": "SimpleVQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 14, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.817, "display_multiplier": 1, "model_count": 14, "model_count_basis": "source_model_id", "numeric_count": 14, "raw_max": 0.817, "raw_min": 0.354, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:simplevqa:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500"}, "unit": null}, "slug": "llm-stats-simplevqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SIQA is a social interaction question answering task, which requires to select the most reasonable behavior based on the given scenario and three possible subsequent behaviors. This task is designed to test the model's knowledge in social commonsense. This dataset consists of 38,963 training samples, 1,951 development samples and 1,960 test samples, all on English text. SIQA 是一个社会交互问答任务，要求根据给定的场景和三个可能的后续行为，选择最合理的行为。这个任务是为了测试模型在社会常识方面的知识。这个数据集包含了 38963 个训练样本，1951 个开发样本和 1960 个测试样本，所有的文本都是英文文本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:533", "languages": ["English"], "modality": null, "name": "SIQA", "openness": "unknown", "publisher": null, "released": "2019-09-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-533-siqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SIQA", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric is the attack success rate; lower is better.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:siren-agentdojo-attack-success:muse-glimmer-30b", "reported_at": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:siren-agentdojo-attack-success", "languages": [], "modality": "text", "name": "Siren AgentDojo Attack Success Rate", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 28.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.284, "raw_min": 0.284, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:siren-agentdojo-attack-success:muse-glimmer-30b", "reported_date": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500"}, "unit": null}, "slug": "llm-stats-siren-agentdojo-attack-success", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500", "unit": null}, {"aliases": [], "categories": ["safety", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric reports utility on the assigned tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:siren-agentdojo-utility:muse-glimmer-30b", "reported_at": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:siren-agentdojo-utility", "languages": [], "modality": "text", "name": "Siren AgentDojo Utility", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 94.19999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.942, "raw_min": 0.942, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:siren-agentdojo-utility:muse-glimmer-30b", "reported_date": "2026-08-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500"}, "unit": null}, "slug": "llm-stats-siren-agentdojo-utility", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SkillsBench evaluates coding agents on self-contained programming tasks, measuring practical engineering skills across diverse software development scenarios.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:skillsbench:qwen3.6-plus", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:skillsbench", "languages": [], "modality": "text", "name": "SkillsBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.19999999999999, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.702, "raw_min": 0.287, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:skillsbench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500"}, "unit": null}, "slug": "llm-stats-skillsbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500", "unit": null}, {"aliases": ["SkillsBench", "SkillsBench V1", "Skills Bench"], "categories": ["agent"], "collected_at": null, "description": "Measures whether curated Agent Skills raise agent pass rates, not raw model capability; the meaningful figure is the paired no-Skills versus curated-Skills gap. The release date follows the paper's first submission.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000skillsbench\u0000tencent_hy4_preview\u0000skillsbench_v1\u0000SkillsBench V1\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000skillsbench\u0000tencent_hy4_preview\u0000skillsbench_v1\u0000SkillsBench V1\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:skillsbench", "languages": [], "modality": null, "name": "SkillsBench", "openness": "unknown", "publisher": null, "released": "2026-02-13", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:skillsbench", "source_url": "https://skillsbench.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.9, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 62.9, "raw_min": 62.9, "source_reference": {"obs_id": "curated\u0000skillsbench\u0000tencent_hy4_preview\u0000skillsbench_v1\u0000SkillsBench V1\u0000Hy4 preview", "observation_id": "curated\u0000skillsbench\u0000tencent_hy4_preview\u0000skillsbench_v1\u0000SkillsBench V1\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "skillsbench", "source": "model_reports", "source_url": "https://skillsbench.ai/", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "image_to_text", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:slakevqa:medgemma-4b-it", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:slakevqa", "languages": [], "modality": "multimodal", "name": "SlakeVQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.6, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.816, "raw_min": 0.623, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:slakevqa:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500"}, "unit": null}, "slug": "llm-stats-slakevqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "Capabilities of on-device LLMs", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SmartBench is the first benchmark specifically designed to evaluate the capabilities of on-device large language models in smartphone scenarios. SmartBench 是首个面向智能手机终端大模型能力评估的基准，基于手机厂商提供的功能将其划分为五类共20项任务，涵盖文本摘要、问答、信息抽取、内容创作和通知管理等场景。它提供高质量数据集与定制化评估标准，旨在推动终端大模型在移动应用中的标准化评估与发展。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2217", "languages": ["Chinese", "Multilingual"], "modality": "multimodal", "name": "SmartBench", "openness": "unknown", "publisher": "VIVO", "released": "2025-09-26", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2217-smartbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SmartBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "医学", "科学智能", "AI for Science", "跨模态推理", "Cross-modal Reasoning", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SMMILE is a benchmark for evaluating multimodal large language models (MLLMs) in medical in-context learning tasks, constructed by medical experts. SMMILE 是一个由医学专家主导构建的多模态医疗上下文学习评测基准，旨在评估多模态大语言模型（MLLMs）在医学任务中的上下文学习能力。该基准包含111个问题（共517个图文问答三元组），涵盖6个医学专科和13种影像模态，另提供扩展版本 SMMILE++，包含1038个变换问题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2029", "languages": [], "modality": "multimodal", "name": "SMMILE", "openness": "unknown", "publisher": "ETH Zurich , Stanford University , Lund University , etc.", "released": "2025-06-26", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2029-smmile", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SMMILE", "unit": null}, {"aliases": [], "categories": ["psychology", "reasoning", "creativity"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The first large-scale benchmark for commonsense reasoning about social situations. Contains 38,000 multiple choice questions probing emotional and social intelligence in everyday situations, testing commonsense understanding of social interactions and theory of mind reasoning about the implied emotions and behavior of others.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:social-iqa:gemma-2-27b-it", "reported_at": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:social-iqa", "languages": [], "modality": "text", "name": "Social IQa", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.0, "display_multiplier": 100, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.78, "raw_min": 0.488, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:social-iqa:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500"}, "unit": null}, "slug": "llm-stats-social-iqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "理解", "Understanding", "3D Spatial Reasoning", "Spatial Intelligence", "物理智能", "Embodied AI", "逻辑推理", "图像理解", "Image Understanding", "空间理解", "Spatial Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Spatial457 systematically introduces four key capabilities—multi-object understanding, 2D and 3D localization, and 3D orientation—across five difficulty levels and seven question types, from simple object recognition to complex 6DoF spatial reasoning tasks. Spatial457 聚焦于空间推理的四项核心能力：多物体理解、二维位置识别、三维位置识别，以及三维朝向判断。这些能力对现实世界中复杂场景的认知和理解至关重要。我们构建了一种层级递进的评估结构，将问题划分为 7 类问题类型，覆盖从基础的单物体识别任务，到我们首次提出的更具挑战性的 6DoF 空间推理任务。整个数据集按 5 个难度等级组织，帮助系统性地评估模型在不同层次空间理解任务中的表现。\n\nSpatial457 不仅提升了空间推理评测的覆盖面和挑战性，也为后续研究提供了统一的基准，有助于推动多模态模型在真实世界空间认知任务中的发展。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2092", "languages": [], "modality": "multimodal", "name": "Spatial457", "openness": "open", "publisher": "Johns Hopkins University", "released": "2025-04-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2092-spatial457", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Spatial457", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A large-scale, complex and cross-domain semantic parsing and text-to-SQL dataset annotated by 11 college students. Contains 10,181 questions and 5,693 unique complex SQL queries on 200 databases with multiple tables, covering 138 different domains. Requires models to generalize to both new SQL queries and new database schemas, making it distinct from previous semantic parsing tasks that use single databases.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:spider:codestral-22b", "reported_at": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:spider", "languages": [], "modality": "text", "name": "Spider", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.635, "raw_min": 0.311, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:spider:codestral-22b", "reported_date": "2024-05-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500"}, "unit": null}, "slug": "llm-stats-spider", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "大语言模型", "LLM", "代码工程", "Code", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Spider is a large-scale complex and cross-domain semantic parsing and text-to-SQL dataset annotated by 11 Yale students. The goal of the Spider challenge is to develop natural language interfaces to cross-domain databases. Spider 是一个大规模、复杂且跨领域的语义解析和文本到 SQL 数据集。Spider 挑战的目标是开发自然语言接口以访问跨领域数据库。该数据集包含 10,181 个问题和 5,693 个独特的复杂 SQL 查询，涵盖 200 个包含多个表的数据库，涉及 138 个不同的领域。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1133", "languages": [], "modality": null, "name": "Spider", "openness": "unknown", "publisher": "Department of Computer Science, Yale University", "released": "2019-02-02", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1133-spider", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Spider", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "NeurIPS 2024", "任务执行", "Task Execution", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Spider2-V is the first multimodal agent benchmark focusing on professional data science and engineering workflows, featuring 494 real-world tasks in authentic computer environments and incorporating 20 enterprise-level professional applications. Spider2-V是第一个专注于专业数据科学和工程工作流程的多模态代理基准测试，整合了20 个企业级专业应用程序，包含来自真实计算机环境的494 个真实任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1270", "languages": [], "modality": null, "name": "Spider2-V", "openness": "unknown", "publisher": "The University of Hong Kong", "released": "2024-07-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1270-spider2-v", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Spider2-V", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "NAACL 2024", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SportQA aims to evaluate LLMs in the context of sports understanding. SportQA encompasses over 70,000 multiple-choice questions across three distinct difficulty levels, each targeting different aspects of sports knowledge from basic historical facts to intricate, scenariobased reasoning tasks. SportQA 专门用于评估大型语言模型（LLMs）在体育理解方面的能力。SportQA 包含超过 70,000 道多项选择题，分为三个不同的难度级别，针对从基本历史事实到复杂情境推理任务的各种体育知识。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1152", "languages": [], "modality": null, "name": "SportQA", "openness": "unknown", "publisher": "University of California", "released": "2024-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1152-sportqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SportQA", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "NeurIPS 2024", "大语言模型", "LLM", "代码工程", "Code", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SpreadsheetBench is a challenging spreadsheet manipulation benchmark built from 912 real questions gathered from online Excel forums. SpreadsheetBench是一个具有挑战性的电子表格操作基准测试，包含912个来自在线Excel论坛的真实问题", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1267", "languages": [], "modality": null, "name": "SpreadsheetBench", "openness": "unknown", "publisher": "Renmin University of China", "released": "2024-06-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1267-spreadsheetbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SpreadsheetBench", "unit": null}, {"aliases": [], "categories": ["productivity", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SpreadsheetBench 2 evaluates office automation agents on spreadsheet analysis, reasoning, and manipulation tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:spreadsheetbench-2:kimi-k3", "reported_at": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:spreadsheetbench-2", "languages": [], "modality": "text", "name": "SpreadsheetBench 2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 34.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.348, "raw_min": 0.348, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:spreadsheetbench-2:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500"}, "unit": null}, "slug": "llm-stats-spreadsheetbench-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500", "unit": null}, {"aliases": [], "categories": ["productivity", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SpreadSheetBench-v1 evaluates office automation agents on spreadsheet reasoning and manipulation tasks, measuring the ability to analyze, transform, and operate on spreadsheet data through tools.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:spreadsheetbench-v1:qwen3.7-max", "reported_at": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:spreadsheetbench-v1", "languages": [], "modality": "text", "name": "SpreadSheetBench-v1", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.35, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.8935, "raw_min": 0.863, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:spreadsheetbench-v1:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500"}, "unit": null}, "slug": "llm-stats-spreadsheetbench-v1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500", "unit": null}, {"aliases": [], "categories": ["long_context", "summarization", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SQuALITY (Summarization-format QUestion Answering with Long Input Texts, Yes!) is a long-document summarization dataset built by hiring highly-qualified contractors to read public-domain short stories (3000-6000 words) and write original summaries from scratch. Each document has five summaries: one overview and four question-focused summaries. Designed to address limitations in existing summarization datasets by providing high-quality, faithful summaries.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:squality:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:squality", "languages": [], "modality": "text", "name": "SQuALITY", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 24.3, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.243, "raw_min": 0.188, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:squality:phi-3.5-mini-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500"}, "unit": null}, "slug": "llm-stats-squality", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "ACL 2024", "大语言模型", "LLM", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "StableToolBench, a benchmark evolving from ToolBench, proposing a virtual API server and stable evaluation system. The virtual API server contains a caching system and API simulators which are complementary to alleviate the change in API status. StableToolBench 是一个从 ToolBench 发展而来的基准，提出了一个虚拟 API 服务器和稳定的评估系统。虚拟 API 服务器包含一个缓存系统和 API 模拟器，这些组件相辅相成，以缓解 API 状态变化带来的影响。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1084", "languages": [], "modality": null, "name": "StableToolBench", "openness": "unknown", "publisher": "THUNLP-MT", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1084-stabletoolbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StableToolBench", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "STARK is a comprehensive benchmark designed to systematically evaluate large language models (LLMs) and large reasoning models (LRMs) on spatiotemporal reasoning tasks, particularly for applications in cyber-physical systems (CPS) such as robotics, autonomous vehicles, and smart city infrastructure. STARK 是一个全面的基准测试套件，旨在系统评估大语言模型（LLMs）和大推理模型（LRMs）在时空推理任务中的表现，特别是在网络物理系统（CPS）中的应用，如机器人、自动驾驶和智能城市基础设施。该基准包含 26 种不同的时空任务，涵盖状态估计、时空关系推理和世界知识感知推理三个层次。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1845", "languages": [], "modality": "multimodal", "name": "STARK_10k", "openness": "unknown", "publisher": "Department of Electrical and Computer Engineering, UCLA", "released": "2025-05-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1845-stark-10k", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/STARK_10k", "unit": null}, {"aliases": [], "categories": ["math", "multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive multimodal benchmark dataset with 448 skills and 1,073,146 questions spanning all STEM subjects (Science, Technology, Engineering, Mathematics), designed to test neural models' vision-language STEM skills based on K-12 curriculum. Unlike existing datasets that focus on expert-level ability, this dataset includes fundamental skills designed around educational standards.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:stem:qwen-2.5-coder-7b-instruct", "reported_at": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:stem", "languages": [], "modality": "multimodal", "name": "STEM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 34.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.34, "raw_min": 0.34, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:stem:qwen-2.5-coder-7b-instruct", "reported_date": "2024-09-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500"}, "unit": null}, "slug": "llm-stats-stem", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "逻辑推理", "Reasoning", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "STRATEGYQA is a question answering (QA) benchmark where the required reasoning steps are implicit in the question, and should be inferred using a strategy STRATEGYQA 是一个问答基准，其中所需的推理步骤在问题中是隐含的，可以通过策略进行推断。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1107", "languages": [], "modality": null, "name": "StrategyQA", "openness": "unknown", "publisher": "Allen Institute for AI", "released": "2021-01-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1107-strategyqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StrategyQA", "unit": null}, {"aliases": [], "categories": ["指令跟随", "Instruct", "大语言模型", "LLM", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "StructFlowBench is a structurally annotated multi-turn benchmark that leverages a structure-driven generation paradigm to enhance the simulation of complex dialogue scenarios. StructFlowBench，这是一个包含155条数据的结构化标注多轮基准，它利用结构驱动生成范式来增强复杂对话场景的模拟。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1542", "languages": [], "modality": null, "name": "StructFlowBench", "openness": "open", "publisher": "Jilin University;University of North Carolina at Chapel Hill;etc.", "released": "2025-02-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1542-structflowbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StructFlowBench", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "科学智能", "AI for Science", "科学推理", "Scientific Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "StructTokenBench is a benchmark framework designed to comprehensively evaluate the quality and efficiency of protein structure tokenization methods, particularly focusing on fine-grained local substructures. StructTokenBench 是一个评估蛋白质结构标记化方法质量效率的基准。它关注细粒度局部子结构，评测维度包括下游有效性、敏感性、独特性和码本利用效率。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2082", "languages": [], "modality": null, "name": "StructTokenBench", "openness": "unknown", "publisher": "Mila - Quebec AI Institute , University of Montreal , Amazon , George Mason Univ", "released": "2025-02-28", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the structure-tokenizer evaluation framework.", "source_key": "opencompass:2082", "source_url": "https://arxiv.org/abs/2503.00089"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2082-structtokenbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StructTokenBench", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "ACL 2024", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "STUDENTEVAL contains 1,749 prompts written by 80 students who have only completed one introductory Python course. STUDENTEVAL contains numerous non-expert prompts describing the same problem, enabling exploration of key factors in prompt success. StudentEval 包含 1,749 个由 80 名仅完成一门入门 Python 课程的学生撰写的提示。StudentEval 中包含许多非专家提示，描述相同的问题，使得探索提示成功的关键因素成为可能。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1082", "languages": [], "modality": null, "name": "StudentEval", "openness": "unknown", "publisher": "Oberlin College", "released": "2024-08-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1082-studenteval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StudentEval", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "语言", "Language", "NLP", "Writing Style Transformation", "Prompt Recovery", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "A Benchmark Dataset for Prompt Recovery in Writing Style Transformation. 基于写作风格转换的提示词恢复的评测集", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1744", "languages": ["English"], "modality": null, "name": "StyleRec", "openness": "unknown", "publisher": null, "released": "2025-04-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1744-stylerec", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StyleRec", "unit": null}, {"aliases": [], "categories": ["long_context", "summarization"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SummScreenFD is the ForeverDreaming subset of the SummScreen dataset for abstractive screenplay summarization, comprising pairs of TV series transcripts and human-written recaps from 88 different shows. The dataset provides a challenging testbed for abstractive summarization where plot details are often expressed indirectly in character dialogues and scattered across the entirety of the transcript, requiring models to find and integrate these details to form succinct plot descriptions.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:summscreenfd:phi-3.5-mini-instruct", "reported_at": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:summscreenfd", "languages": [], "modality": "text", "name": "SummScreenFD", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 16.900000000000002, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.169, "raw_min": 0.16, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:summscreenfd:phi-3.5-moe-instruct", "reported_date": "2024-08-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500"}, "unit": null}, "slug": "llm-stats-summscreenfd", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500", "unit": null}, {"aliases": [], "categories": ["spatial_reasoning", "3d", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SUNRGBD evaluates RGB-D scene understanding and 3D grounding capabilities.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:sunrgbd:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:sunrgbd", "languages": [], "modality": "image", "name": "SUNRGBD", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.362, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.362, "raw_min": 0.334, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:sunrgbd:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500"}, "unit": null}, "slug": "llm-stats-sunrgbd", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "knowledge", "chemistry"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SuperChem is a benchmark of advanced chemistry problems requiring expert-level domain knowledge and reasoning.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:superchem:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:superchem", "languages": [], "modality": "text", "name": "SuperChem", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.8, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.598, "raw_min": 0.549, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:superchem:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500"}, "unit": null}, "slug": "llm-stats-superchem", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500", "unit": null}, {"aliases": [], "categories": ["science"], "collected_at": null, "description": "Multimodal chemical reasoning; image and formula parsing quality moves the reported figure.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000superchem\u0000tencent_hy4_preview\u0000superchem\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000superchem\u0000tencent_hy4_preview\u0000superchem\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:superchem", "languages": [], "modality": null, "name": "SUPERChem", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 66.4, "raw_min": 66.4, "source_reference": {"obs_id": "curated\u0000superchem\u0000tencent_hy4_preview\u0000superchem\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000superchem\u0000tencent_hy4_preview\u0000superchem\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "superchem", "source": "model_reports", "source_url": "https://github.com/catalystforyou/SUPERChem_eval", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "language", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SuperGLUE is a new benchmark styled after GLUE with a new set of more difficult language understanding tasks, improved resources, and a new public leaderboard. It includes 8 primary tasks: BoolQ (Boolean Questions), CB (CommitmentBank), COPA (Choice of Plausible Alternatives), MultiRC (Multi-Sentence Reading Comprehension), ReCoRD (Reading Comprehension with Commonsense Reasoning), RTE (Recognizing Textual Entailment), WiC (Word-in-Context), and WSC (Winograd Schema Challenge). The benchmark evaluates diverse language understanding capabilities including reading comprehension, commonsense reasoning, causal reasoning, coreference resolution, textual entailment, and word sense disambiguation across multiple domains.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:superglue:o1-mini", "reported_at": "2024-09-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:superglue", "languages": [], "modality": "text", "name": "SuperGLUE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.75, "raw_min": 0.75, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:superglue:o1-mini", "reported_date": "2024-09-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500"}, "unit": null}, "slug": "llm-stats-superglue", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "math", "physics", "reasoning", "finance", "general", "healthcare", "chemistry", "economics"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SuperGPQA is a comprehensive benchmark that evaluates large language models across 285 graduate-level academic disciplines. The benchmark contains 25,957 questions covering 13 broad disciplinary areas including Engineering, Medicine, Science, and Law, with specialized fields in light industry, agriculture, and service-oriented domains. It employs a Human-LLM collaborative filtering mechanism with over 80 expert annotators to create challenging questions that assess graduate-level knowledge and reasoning capabilities.", "evidence_summary": {"document_count": 1, "model_count": 34, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:supergpqa:qwen3-235b-a22b", "reported_at": "2025-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:supergpqa", "languages": [], "modality": "text", "name": "SuperGPQA", "openness": "restricted", "publisher": "M-A-P,ByteDance.Inc, 2077.AI", "released": "2025-02-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 34, "score_direction": "higher_is_better", "score_summary": {"display_max": 73.6, "display_multiplier": 100, "model_count": 34, "model_count_basis": "source_model_id", "numeric_count": 34, "raw_max": 0.736, "raw_min": 0.213, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:supergpqa:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500"}, "unit": null}, "slug": "llm-stats-supergpqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500", "unit": null}, {"aliases": ["SuperGPQA", "Super-GPQA"], "categories": ["knowledge"], "collected_at": null, "description": "285 graduate disciplines; breadth comes at the cost of per-field sample size.", "evidence_summary": {"document_count": 2, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000super_gpqa\u0000qwen3_5_model_card\u0000super_gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000super_gpqa\u0000qwen3_5_model_card\u0000super_gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:super_gpqa", "languages": [], "modality": null, "name": "SuperGPQA", "openness": "unknown", "publisher": null, "released": "2025-02-20", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:super_gpqa", "source_url": "https://arxiv.org/abs/2502.14739"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.4, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 70.4, "raw_min": 70.4, "source_reference": {"obs_id": "curated\u0000super_gpqa\u0000qwen3_5_model_card\u0000super_gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "observation_id": "curated\u0000super_gpqa\u0000qwen3_5_model_card\u0000super_gpqa\u0000thinking (temp 0.6, top_p 0.95, top_k 20)\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "reported_date": "2026-02-16", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "unit": "percent"}, "slug": "super_gpqa", "source": "model_reports", "source_url": "https://arxiv.org/abs/2502.14739", "unit": "percent"}, {"aliases": [], "categories": ["学科", "Examination", "大语言模型", "LLM", "知识储备", "Knowledge", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SuperGPQA, a comprehensive benchmark designed to evaluate the knowledge and reasoning abilities of Large Language Models (LLMs) across 285 graduate-level disciplines. SuperGPQA features at least 50 questions per discipline, covering a broad spectrum of graduate-level topics. SuperGPQA，这是一个旨在评估大型语言模型在 285 个研究生学科领域的知识和推理能力的全面基准。SuperGPQA 每个学科至少包含 50 个问题，涵盖广泛的硕士研究生学科主题。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1538", "languages": [], "modality": null, "name": "SuperGPQA", "openness": "restricted", "publisher": "M-A-P,ByteDance.Inc, 2077.AI", "released": "2025-02-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1538-supergpqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SuperGPQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SURDS is a benchmark for spatial understanding and reasoning in autonomous-driving scenes.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:surds:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:surds", "languages": [], "modality": "multimodal", "name": "SURDS", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.2, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.772, "raw_min": 0.772, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:surds:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500"}, "unit": null}, "slug": "llm-stats-surds", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500", "unit": null}, {"aliases": [], "categories": ["数学", "Math", "大语言模型", "LLM", "数理能力", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SVAMP includes one-unknown arithmetic word problems with grade level up to 4 by applying simple variations over word problems in an existing dataset. SVAMP further highlights the brittle nature of existing models when trained on these benchmark datasets. SVAMP 是一个包含算术文字问题的数据集，最高适用于四年级的学生，是通过对现有数据集中的文字问题应用简单变体而生成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1113", "languages": [], "modality": null, "name": "SVAMP", "openness": "open", "publisher": "Microsoft Research India", "released": "2021-04-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1113-svamp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SVAMP", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SVG-Bench is an internal benchmark that comprehensively evaluates SVG generation performance. It accepts text and image inputs across build-from-scratch and edit-based tasks, using a VLM to verify rendering accuracy of the generated outputs.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:svg-bench:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:svg-bench", "languages": [], "modality": "multimodal", "name": "SVG-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.637, "raw_min": 0.637, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:svg-bench:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-svg-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500", "unit": null}, {"aliases": ["SWE Atlas", "SWE-Atlas"], "categories": ["coding_agent"], "collected_at": null, "description": "Reports three splits (Codebase Q&A, Test Writing, Refactoring) as separate figures; the split is part of the instrument.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_atlas\u0000tencent_hy4_preview\u0000swe_atlas_codebase_qa\u0000Codebase Q&A split\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_atlas\u0000tencent_hy4_preview\u0000swe_atlas_codebase_qa\u0000Codebase Q&A split\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:swe_atlas", "languages": [], "modality": null, "name": "SWE Atlas", "openness": "unknown", "publisher": null, "released": "2026-05-08", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:swe_atlas", "source_url": "https://github.com/scaleapi/SWE-Atlas"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.0, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 64.0, "raw_min": 53.3, "source_reference": {"obs_id": "curated\u0000swe_atlas\u0000tencent_hy4_preview\u0000swe_atlas_codebase_qa\u0000Codebase Q&A split\u0000Hy4 preview", "observation_id": "curated\u0000swe_atlas\u0000tencent_hy4_preview\u0000swe_atlas_codebase_qa\u0000Codebase Q&A split\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "swe_atlas", "source": "model_reports", "source_url": "https://github.com/scaleapi/SWE-Atlas", "unit": "percent"}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE Atlas - Codebase QnA evaluates a model's ability to answer questions about real codebases, measuring repository-level comprehension and the ability to reason about code structure, behavior, and intent across an entire project.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-atlas-codebase-qna:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-atlas-codebase-qna", "languages": [], "modality": "text", "name": "SWE Atlas - Codebase QnA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 46.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.462, "raw_min": 0.379, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-atlas-codebase-qna:laguna-s-2.1", "reported_date": "2026-07-21", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-atlas-codebase-qna", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE Atlas - Test Writing evaluates a model's ability to author meaningful tests for real-world software projects, measuring how well agents can understand code and produce correct, useful test coverage.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-atlas-test-writing:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-atlas-test-writing", "languages": [], "modality": "text", "name": "SWE Atlas - Test Writing", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 30.830000000000002, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.3083, "raw_min": 0.3083, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-atlas-test-writing:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-atlas-test-writing", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-Atlas is a software engineering benchmark focused on debugging, evaluating a model's ability to localize and fix bugs in real-world codebases.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-atlas:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-atlas", "languages": [], "modality": "text", "name": "SWE-Atlas", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 35.199999999999996, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.352, "raw_min": 0.306, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-atlas:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-atlas", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multilingual benchmark for issue resolving in software engineering that covers Java, TypeScript, JavaScript, Go, Rust, C, and C++. Contains 1,632 high-quality instances carefully annotated from 2,456 candidates by 68 expert annotators, designed to evaluate Large Language Models across diverse software ecosystems beyond Python.", "evidence_summary": {"document_count": 1, "model_count": 38, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-bench-multilingual:deepseek-v3.1", "reported_at": "2025-01-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "llm-stats:swe-bench-multilingual", "languages": [], "modality": "text", "name": "SWE-bench Multilingual", "openness": "open", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 38, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.3, "display_multiplier": 100, "model_count": 38, "model_count_basis": "source_model_id", "numeric_count": 38, "raw_max": 0.873, "raw_min": 0.305, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-bench-multilingual:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-bench-multilingual", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500", "unit": null}, {"aliases": ["SWE-bench Multilingual", "SWE Multilingual"], "categories": ["coding_agent"], "collected_at": null, "description": "Per-language resolution rates differ widely, so an average hides which languages the model actually handles.", "evidence_summary": {"document_count": 5, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_bench_multilingual\u0000zai_glm_5_model_card\u0000swe_bench_multilingual\u0000OpenHands, tailored instruction prompt, temp 0.7, top_p 0.95, 200K context\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_bench_multilingual\u0000zai_glm_5_model_card\u0000swe_bench_multilingual\u0000OpenHands, tailored instruction prompt, temp 0.7, top_p 0.95, 200K context\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:swe_bench_multilingual", "languages": [], "modality": null, "name": "SWE-bench Multilingual", "openness": "unknown", "publisher": null, "released": "2025-04-22", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:swe_bench_multilingual", "source_url": "https://www.swebench.com/multilingual.html"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.9, "display_multiplier": 1, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 82.9, "raw_min": 69.3, "source_reference": {"obs_id": "curated\u0000swe_bench_multilingual\u0000tencent_hy4_preview\u0000swe_bench_multilingual\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000swe_bench_multilingual\u0000tencent_hy4_preview\u0000swe_bench_multilingual\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "swe_bench_multilingual", "source": "model_reports", "source_url": "https://www.swebench.com/multilingual.html", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "reasoning", "agents", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-Bench Multimodal extends SWE-Bench to evaluate language models on software engineering tasks that involve visual inputs such as screenshots, UI mockups, and diagrams alongside code understanding.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-bench-multimodal:claude-mythos-preview", "reported_at": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-bench-multimodal", "languages": [], "modality": "multimodal", "name": "SWE-Bench Multimodal", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 59.0, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.59, "raw_min": 0.281, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-bench-multimodal:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-bench-multimodal", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-Bench Pro is an advanced version of SWE-Bench that evaluates language models on complex, real-world software engineering tasks requiring extended reasoning and multi-step problem solving.", "evidence_summary": {"document_count": 1, "model_count": 50, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-bench-pro:gpt-5.2-codex", "reported_at": "2026-01-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "llm-stats:swe-bench-pro", "languages": [], "modality": "text", "name": "SWE-Bench Pro", "openness": "restricted", "publisher": "Scale AI", "released": "2025-09-05", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 50, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.0, "display_multiplier": 100, "model_count": 50, "model_count_basis": "source_model_id", "numeric_count": 50, "raw_max": 0.8, "raw_min": 0.402, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-bench-pro:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-bench-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500", "unit": null}, {"aliases": ["SWE-bench Pro", "SWE Pro", "SWE-Bench Pro (Public)"], "categories": ["coding_agent"], "collected_at": null, "description": "Harness and split version must be recorded with any score.", "evidence_summary": {"document_count": 10, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_bench_pro\u0000google_gemini_3_1_pro_model_card\u0000swe_bench_pro\u0000Thinking (High), Single attempt\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "first_score_reported_at": "2026-02-19", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_bench_pro\u0000google_gemini_3_1_pro_model_card\u0000swe_bench_pro\u0000Thinking (High), Single attempt\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:swe_bench_pro", "languages": [], "modality": null, "name": "SWE-bench Pro", "openness": "unknown", "publisher": null, "released": "2025-09-23", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:swe_bench_pro", "source_url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 79.2, "display_multiplier": 1, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 79.2, "raw_min": 52.1, "source_reference": {"obs_id": "curated\u0000swe_bench_pro\u0000anthropic_claude_opus_5_system_card\u0000swe_bench_pro\u0000adaptive thinking, max effort, averaged over 5 trials\u0000Claude Opus 5", "observation_id": "curated\u0000swe_bench_pro\u0000anthropic_claude_opus_5_system_card\u0000swe_bench_pro\u0000adaptive thinking, max effort, averaged over 5 trials\u0000Claude Opus 5", "reported_at": "2026-07-24", "reported_date": "2026-07-24", "source_id": "anthropic_claude_opus_5_system_card", "source_url": "https://www.anthropic.com/news/claude-opus-5"}, "unit": "percent"}, "slug": "swe_bench_pro", "source": "model_reports", "source_url": "https://github.com/scaleapi/SWE-bench_Pro-os", "unit": "percent"}, {"aliases": ["SWE-bench Science", "SWE-Bench-Science", "SWE Bench Science"], "categories": ["coding_agent"], "collected_at": null, "description": "Repository-level scientific software engineering benchmark from the OpenMOSS team (Fudan): 119 tasks across 98 GitHub repositories in 20 scientific domains, organized into Issue-driven, Expert-exploratory, and Engineering-integration paradigms. Best reported pass@1 is below 50% (Claude Code with Opus-5 max), and the paper identifies four recurring failure mechanisms, so scores are highly sensitive to scientific-domain knowledge and tool access, not only to coding ability.", "evidence_summary": {"document_count": null, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:swe_bench_science", "languages": [], "modality": null, "name": "SWE-bench Science", "openness": "unknown", "publisher": null, "released": "2026-08-20", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:swe_bench_science", "source_url": "https://arxiv.org/abs/2608.19799"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "swe_bench_science", "source": "model_reports", "source_url": "https://arxiv.org/abs/2608.19799", "unit": null}, {"aliases": [], "categories": ["reasoning", "frontend_development", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A verified subset of 500 software engineering problems from real GitHub issues, validated by human annotators for evaluating language models' ability to resolve real-world coding issues by generating patches for Python codebases.", "evidence_summary": {"document_count": 1, "model_count": 111, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-bench-verified:deepseek-v2.5", "reported_at": "2024-05-08", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "llm-stats:swe-bench-verified", "languages": [], "modality": "text", "name": "SWE-Bench Verified", "openness": "restricted", "publisher": "OpenAI", "released": "2024-08-13", "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 111, "score_direction": "higher_is_better", "score_summary": {"display_max": 95.0, "display_multiplier": 100, "model_count": 111, "model_count_basis": "source_model_id", "numeric_count": 111, "raw_max": 0.95, "raw_min": 0.087, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-bench-verified:claude-fable-5", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-bench-verified", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500", "unit": null}, {"aliases": ["SWE-bench Verified", "SWE-bench-Verified", "SWE Bench Verified", "SWE Verified"], "categories": ["coding_agent"], "collected_at": null, "description": "Scaffold, tool permissions and time limit are part of the result; vendors run their own harnesses.", "evidence_summary": {"document_count": 18, "model_count": 12, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_bench_verified\u0000deepseek_v3_report\u0000swe_bench_verified\u0000Resolved, vendor harness\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "first_score_reported_at": "2024-12-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_bench_verified\u0000deepseek_v3_report\u0000swe_bench_verified\u0000Resolved, vendor harness\u0000DeepSeek-V3", "reported_at": "2024-12-27", "source_url": "https://arxiv.org/abs/2412.19437"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:swe_bench_verified", "languages": [], "modality": null, "name": "SWE-bench Verified", "openness": "unknown", "publisher": null, "released": "2024-08-13", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:swe_bench_verified", "source_url": "https://openai.com/index/introducing-swe-bench-verified/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 16, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.0, "display_multiplier": 1, "model_count": 12, "model_count_basis": "source_model_id", "numeric_count": 16, "raw_max": 96.0, "raw_min": 42.0, "source_reference": {"obs_id": "curated\u0000swe_bench_verified\u0000anthropic_claude_opus_5_system_card\u0000swe_bench_verified\u0000adaptive thinking, max effort, averaged over 5 trials\u0000Claude Opus 5", "observation_id": "curated\u0000swe_bench_verified\u0000anthropic_claude_opus_5_system_card\u0000swe_bench_verified\u0000adaptive thinking, max effort, averaged over 5 trials\u0000Claude Opus 5", "reported_at": "2026-07-24", "reported_date": "2026-07-24", "source_id": "anthropic_claude_opus_5_system_card", "source_url": "https://www.anthropic.com/news/claude-opus-5"}, "unit": "percent"}, "slug": "swe_bench_verified", "source": "model_reports", "source_url": "https://openai.com/index/introducing-swe-bench-verified/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-bench Verified is a human-filtered subset of 500 software engineering problems drawn from real GitHub issues across 12 popular Python repositories. Given a codebase and an issue description, language models are tasked with generating patches that resolve the described problems. This benchmark evaluates AI's real-world agentic coding skills by requiring models to navigate complex codebases, understand software engineering problems, and coordinate changes across multiple functions, classes, and files to fix well-defined issues with clear descriptions.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-bench-verified-(agentic-coding):kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-bench-verified-(agentic-coding)", "languages": [], "modality": "text", "name": "SWE-bench Verified (Agentic Coding)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.2, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.772, "raw_min": 0.658, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-bench-verified-(agentic-coding):claude-sonnet-4-5-20250929", "reported_date": "2025-09-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-bench-verified-agentic-coding", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A human-validated subset of SWE-bench that evaluates language models' ability to resolve real-world GitHub issues using an agentless approach. The benchmark tests models on software engineering problems requiring understanding and coordinating changes across multiple functions, classes, and files simultaneously.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-bench-verified-(agentless):kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-bench-verified-(agentless)", "languages": [], "modality": "text", "name": "SWE-bench Verified (Agentless)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.800000000000004, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.518, "raw_min": 0.357, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-bench-verified-(agentless):kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-bench-verified-agentless", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-bench Verified is a human-validated subset of 500 test samples from the original SWE-bench dataset that evaluates AI systems' ability to automatically resolve real GitHub issues in Python repositories. Given a codebase and issue description, models must edit the code to successfully resolve the problem, requiring understanding and coordination of changes across multiple functions, classes, and files. The Verified version provides more reliable evaluation through manual validation of test samples.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-bench-verified-(multiple-attempts):kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-bench-verified-(multiple-attempts)", "languages": [], "modality": "text", "name": "SWE-bench Verified (Multiple Attempts)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.716, "raw_min": 0.716, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-bench-verified-(multiple-attempts):kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-bench-verified-multiple-attempts", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "智能体", "Agent", "任务执行", "Task Execution", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SWE-bench-Live is a live-updatable benchmark designed for evaluating large language models (LLMs) and agents on real-world software issue resolution tasks. SWE-bench-Live 是一个面向大语言模型（LLMs）和智能体的实时可更新评测基准，专注于真实世界软件开发中的问题修复任务。 该基准从 2024 年以来的 GitHub 活跃仓库中自动收集了 1,319 个问题修复任务，涵盖 93 个项目，并为每个任务提供可复现的 Docker 执行环境。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1984", "languages": [], "modality": null, "name": "SWE-bench-Live", "openness": "unknown", "publisher": "Microsoft", "released": "2025-06-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1984-swe-bench-live", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SWE-bench-Live", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "智能体", "Agent", "任务执行", "Task Execution", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SWE-Factory is a benchmark for evaluating large language models on software issue fixing tasks. SWE-Factory 是一个面向大型语言模型的软件问题修复评测基准，旨在提升构建效率与评估准确性。该基准集成多智能体系统 SWE-Builder 自动搭建任务环境，采用退出码自动评分，并通过 fail2pass 流程验证修复有效性，确保评测可靠。SWE-Factory 覆盖四种语言共 671 个问题，支持高效、自动化、可扩展的 LLM 评估流程。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1989", "languages": [], "modality": null, "name": "SWE-Factory", "openness": "unknown", "publisher": "SunYat-senUniversity , IndependentResearcher , Huawei.", "released": "2025-06-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1989-swe-factory", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SWE-Factory", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-fficiency is an open-source benchmark and workflow that evaluates language models on optimizing the runtime efficiency of real-world software engineering tasks, measuring how well agents can improve code performance autonomously.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-fficiency:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-fficiency", "languages": [], "modality": "text", "name": "SWE-fficiency", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 34.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.348, "raw_min": 0.348, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-fficiency:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-fficiency", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark for evaluating large language models on real-world freelance software engineering tasks from Upwork. Contains over 1,400 tasks valued at $1 million USD total, ranging from $50 bug fixes to $32,000 feature implementations. Includes both independent engineering tasks graded via end-to-end tests and managerial tasks assessed against original engineering managers' choices.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-lancer:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-lancer", "languages": [], "modality": "text", "name": "SWE-Lancer", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 66.3, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.663, "raw_min": 0.18, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-lancer:gpt-5.1-codex", "reported_date": "2025-11-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-lancer", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-Lancer (IC-Diamond subset) is a benchmark of real-world freelance software engineering tasks from Upwork, ranging from $50 bug fixes to $32,000 feature implementations. It evaluates AI models on independent engineering tasks using end-to-end tests triple-verified by experienced software engineers, and includes managerial tasks where models choose between technical implementation proposals.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-lancer-(ic-diamond-subset):gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-lancer-(ic-diamond-subset)", "languages": [], "modality": "text", "name": "SWE-Lancer (IC-Diamond subset)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 100.0, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 1.0, "raw_min": 0.074, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-lancer-(ic-diamond-subset):gpt-5-2025-08-07", "reported_date": "2025-08-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-lancer-ic-diamond-subset", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-Marathon is an ultra-long-horizon software engineering benchmark covering tasks such as building compilers, optimizing kernels, and developing production-grade services. It measures whether agents can sustain quality across extremely long engineering trajectories.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-marathon:glm-5.2", "reported_at": "2026-06-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-marathon", "languages": [], "modality": "text", "name": "SWE-Marathon", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 42.5, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.425, "raw_min": 0.13, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-marathon:glm-5.3", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-marathon", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500", "unit": null}, {"aliases": ["SWE-Marathon", "SWE Marathon"], "categories": ["coding_agent"], "collected_at": null, "description": "Ultra long-horizon software engineering tasks; the score depends on the agent's loop budget as much as on coding skill.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_marathon\u0000tencent_hy4_preview\u0000swe_marathon\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000swe_marathon\u0000tencent_hy4_preview\u0000swe_marathon\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:swe_marathon", "languages": [], "modality": null, "name": "SWE-Marathon", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 31.9, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 31.9, "raw_min": 31.9, "source_reference": {"obs_id": "curated\u0000swe_marathon\u0000tencent_hy4_preview\u0000swe_marathon\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000swe_marathon\u0000tencent_hy4_preview\u0000swe_marathon\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "swe_marathon", "source": "model_reports", "source_url": "https://github.com/abundant-ai/swe-marathon", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "agents", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "SWE-MM evaluates software-engineering agents on repository tasks that require understanding both source code and visual evidence.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-mm:qwen3.8-27b", "reported_at": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-mm", "languages": [], "modality": "multimodal", "name": "SWE-MM", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 38.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.386, "raw_min": 0.386, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-mm:qwen3.8-27b", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-mm", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Software Engineering Performance benchmark measuring code optimization capabilities", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-perf:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-perf", "languages": [], "modality": "text", "name": "SWE-Perf", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 3.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.031, "raw_min": 0.031, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-perf:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-perf", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Software Engineering Review benchmark evaluating code review capabilities", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swe-review:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swe-review", "languages": [], "modality": "text", "name": "SWE-Review", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 8.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.089, "raw_min": 0.089, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swe-review:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500"}, "unit": null}, "slug": "llm-stats-swe-review", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SwiLTra-Bench is a comprehensive multilingual benchmark of over 180K aligned Swiss legal translation pairs comprising laws, headnotes, and press releases across all Swiss languages along with English, designed to evaluate LLM-based translation systems. SwiLTra-Bench是一个包含超过 18 万对对齐的瑞士法律翻译语料库的全面多语言基准，包括所有瑞士语言以及英语的法律、摘要和新闻稿，旨在评估基于LLM的翻译系统。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1608", "languages": ["English", "French", "German", "Italian", "Multilingual"], "modality": null, "name": "SwiLTra-Bench", "openness": "unknown", "publisher": "Harvey, ETHZurich, etc.", "released": "2025-03-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1608-swiltra-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SwiLTra-Bench", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Software Test Benchmark evaluating LLM ability to write tests for software repositories", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:swt-bench:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:swt-bench", "languages": [], "modality": "text", "name": "SWT-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 69.3, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.693, "raw_min": 0.693, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:swt-bench:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-swt-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "agent", "任务执行", "Task Execution", "官方自建", "Official", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "T-Eval evaluates the tool utilization capabilities of LLMs and decomposing them into instruction following, planning, reasoning, retrieval, understanding, and review. T-Eval 评估了 LLM 的工具使用能力，并将其分解为指令遵循、规划、推理、检索、理解和审查等子能力", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:540", "languages": [], "modality": null, "name": "T-Eval", "openness": "open", "publisher": null, "released": "2024-01-15", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-540-t-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/T-Eval", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "t2-bench is a benchmark for evaluating agentic tool use capabilities, measuring how well models can select, sequence, and utilize tools to solve complex tasks. It tests autonomous planning and execution in multi-step scenarios.", "evidence_summary": {"document_count": 1, "model_count": 23, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:t2-bench:gpt-oss-120b-high", "reported_at": "2025-08-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:t2-bench", "languages": [], "modality": "text", "name": "t2-bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "needs_review", "score_count": 23, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.3, "display_multiplier": 100, "model_count": 23, "model_count_basis": "source_model_id", "numeric_count": 23, "raw_max": 0.993, "raw_min": 0.116, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:t2-bench:gemini-3.1-pro-preview", "reported_date": "2026-02-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-t2-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "大语言模型", "LLM", "语言理解", "Comprehension", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TabFac consists of 117,854 manually annotated statements with regard to 16,573 Wikipedia tables, their relations are classified as ENTAILED and REFUTED. TabFac 包含 117,854 条手动标注的语句，涉及 16,573 个维基百科表格，是第一个评估结构化数据上语言推理的数据集，涉及在符号和语言两个方面的混合推理能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1132", "languages": [], "modality": null, "name": "TabFact", "openness": "unknown", "publisher": "Tencent AI Lab", "released": "2020-06-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1132-tabfact", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TabFact", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "理解", "Understanding", "表格问答", "跨语言", "真实世界数据", "大语言模型", "LLM", "逻辑推理", "语言理解", "Comprehension", "OCR与文档理解", "OCR and Document Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TableEval is the first cross-lingual benchmark for tabular question answering, supporting Simplified Chinese, Traditional Chinese, and English. It is designed to evaluate model performance on real-world, complex table understanding tasks across multiple languages. TableEval 是首个支持简体中文、繁体中文和英文的跨语言表格问答基准数据集，旨在系统评估大模型在真实复杂表格理解任务中的表现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2025", "languages": ["English", "Chinese"], "modality": null, "name": "TableEval", "openness": "unknown", "publisher": "北京中科闻歌科技股份有限公司、中国科学院自动化研究所、中国科学院大学", "released": "2025-06-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2025-tableeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TableEval", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "NeurIPS 2024", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TaskBench aims to evaluate the capability of LLMs in task automation, containing 28,271 samples spanning 3 critical stages: task decomposition, tool invocation, and parameter prediction. TaskBench旨在评估LLM在任务自动化方面的能力，包含面向任务分解、工具调用和参数预测三个关键阶段的28271个样本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1164", "languages": [], "modality": null, "name": "TaskBench", "openness": "unknown", "publisher": "Microsoft Research Asia", "released": "2023-11-30", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1164-taskbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TaskBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "τ-bench: A benchmark for tool-agent-user interaction in real-world domains. Tests language agents' ability to interact with users and follow domain-specific rules through dynamic conversations using API tools and policy guidelines across retail and airline domains. Evaluates consistency and reliability of agent behavior over multiple trials.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau-bench:o3-2025-04-16", "reported_at": "2025-04-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tau-bench", "languages": [], "modality": "text", "name": "Tau-bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.2, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.882, "raw_min": 0.63, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau-bench:step-3.5-flash", "reported_date": "2026-02-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-tau-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500", "unit": null}, {"aliases": ["tau-bench", "τ-bench", "tau2-bench", "TAU-bench"], "categories": ["tool_use"], "collected_at": null, "description": "Depends on a simulated user and policy; the simulator model is part of the measurement.", "evidence_summary": {"document_count": 3, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:tau_bench", "languages": [], "modality": null, "name": "tau-bench", "openness": "unknown", "publisher": null, "released": "2024-06-17", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:tau_bench", "source_url": "https://github.com/sierra-research/tau-bench"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "tau_bench", "source": "model_reports", "source_url": "https://github.com/sierra-research/tau-bench", "unit": null}, {"aliases": [], "categories": ["reasoning", "communication", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Part of τ-bench (TAU-bench), a benchmark for Tool-Agent-User interaction in real-world domains. The airline domain evaluates language agents' ability to interact with users through dynamic conversations while following domain-specific rules and using API tools. Agents must handle airline-related tasks and policies reliably.", "evidence_summary": {"document_count": 1, "model_count": 23, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau-bench-airline:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:tau-bench-airline", "languages": [], "modality": "text", "name": "TAU-bench Airline", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 23, "score_direction": "higher_is_better", "score_summary": {"display_max": 70.0, "display_multiplier": 100, "model_count": 23, "model_count_basis": "source_model_id", "numeric_count": 23, "raw_max": 0.7, "raw_min": 0.14, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau-bench-airline:claude-sonnet-4-5-20250929", "reported_date": "2025-09-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500"}, "unit": null}, "slug": "llm-stats-tau-bench-airline", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "communication", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A benchmark for evaluating tool-agent-user interaction in retail environments. Tests language agents' ability to handle dynamic conversations with users while using domain-specific API tools and following policy guidelines. Evaluates agents on tasks like order cancellations, address changes, and order status checks through multi-turn conversations.", "evidence_summary": {"document_count": 1, "model_count": 25, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau-bench-retail:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:tau-bench-retail", "languages": [], "modality": "text", "name": "TAU-bench Retail", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 25, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.2, "display_multiplier": 100, "model_count": 25, "model_count_basis": "source_model_id", "numeric_count": 25, "raw_max": 0.862, "raw_min": 0.226, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau-bench-retail:claude-sonnet-4-5-20250929", "reported_date": "2025-09-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500"}, "unit": null}, "slug": "llm-stats-tau-bench-retail", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "communication", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TAU2 airline domain benchmark for evaluating conversational agents in dual-control environments where both AI agents and users interact with tools in airline customer service scenarios. Tests agent coordination, communication, and ability to guide user actions in tasks like flight booking, modifications, cancellations, and refunds.", "evidence_summary": {"document_count": 1, "model_count": 23, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau2-airline:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:tau2-airline", "languages": [], "modality": "text", "name": "Tau2 Airline", "openness": "restricted", "publisher": "Sierra", "released": "2025-06-09", "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 23, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.5, "display_multiplier": 100, "model_count": 23, "model_count_basis": "source_model_id", "numeric_count": 23, "raw_max": 0.765, "raw_min": 0.44, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau2-airline:longcat-flash-thinking-2601", "reported_date": "2026-01-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500"}, "unit": null}, "slug": "llm-stats-tau2-airline", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "communication", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "τ²-bench retail domain evaluates conversational AI agents in customer service scenarios within a dual-control environment where both agent and user can interact with tools. Tests tool-agent-user interaction, rule adherence, and task consistency in retail customer support contexts.", "evidence_summary": {"document_count": 1, "model_count": 26, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau2-retail:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:tau2-retail", "languages": [], "modality": "text", "name": "Tau2 Retail", "openness": "restricted", "publisher": "Sierra", "released": "2025-06-09", "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 26, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.9, "display_multiplier": 100, "model_count": 26, "model_count_basis": "source_model_id", "numeric_count": 26, "raw_max": 0.919, "raw_min": 0.569, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau2-retail:claude-opus-4-6", "reported_date": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500"}, "unit": null}, "slug": "llm-stats-tau2-retail", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "communication", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "τ²-Bench telecom domain evaluates conversational agents in a dual-control environment modeled as a Dec-POMDP, where both agent and user use tools in shared telecommunications troubleshooting scenarios that test coordination and communication capabilities.", "evidence_summary": {"document_count": 1, "model_count": 35, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau2-telecom:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:tau2-telecom", "languages": [], "modality": "text", "name": "Tau2 Telecom", "openness": "restricted", "publisher": "Sierra", "released": "2025-06-09", "released_reference": null, "repo_kind": "shared_parent", "repo_resolution_status": "resolved", "score_count": 35, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.3, "display_multiplier": 100, "model_count": 35, "model_count_basis": "source_model_id", "numeric_count": 35, "raw_max": 0.993, "raw_min": 0.132, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau2-telecom:claude-opus-4-6", "reported_date": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500"}, "unit": null}, "slug": "llm-stats-tau2-telecom", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500", "unit": null}, {"aliases": ["tau2-bench", "τ²-Bench", "TAU2-Bench", "Tau2", "τ2-bench"], "categories": ["tool_use"], "collected_at": null, "description": "Per-domain scores (Retail, Telecom, Airline) diverge sharply and several cards apply the Airline domain fixes from the Claude Opus 4.5 system card, so an average conceals both the domain mix and the patch level.", "evidence_summary": {"document_count": 6, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000tau2_bench\u0000zai_glm_5_model_card\u0000tau2_bench\u0000Retail and Telecom prompt adjustment; Airline domain fixes per Claude Opus 4.5 system card\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000tau2_bench\u0000zai_glm_5_model_card\u0000tau2_bench\u0000Retail and Telecom prompt adjustment; Airline domain fixes per Claude Opus 4.5 system card\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:tau2_bench", "languages": [], "modality": null, "name": "tau2-bench", "openness": "unknown", "publisher": null, "released": "2025-06-09", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:tau2_bench", "source_url": "https://arxiv.org/abs/2506.07982"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.3, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 99.3, "raw_min": 76.9, "source_reference": {"obs_id": "curated\u0000tau2_bench\u0000google_gemini_3_1_pro_model_card\u0000tau2_bench\u0000Thinking (High), Telecom\u0000Gemini 3.1 Pro", "observation_id": "curated\u0000tau2_bench\u0000google_gemini_3_1_pro_model_card\u0000tau2_bench\u0000Thinking (High), Telecom\u0000Gemini 3.1 Pro", "reported_at": "2026-02-19", "reported_date": "2026-02-19", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"}, "unit": "percent"}, "slug": "tau2_bench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2506.07982", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "τ³-Bench airline domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated airline booking and reservations environment.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau3-airline:mistral-medium-3-5", "reported_at": "2026-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tau3-airline", "languages": [], "modality": "text", "name": "Tau3 Airline", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.72, "raw_min": 0.72, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau3-airline:mistral-medium-3-5", "reported_date": "2026-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500"}, "unit": null}, "slug": "llm-stats-tau3-airline", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "τ³-Bench banking domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated retail banking environment.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau3-banking:mistral-medium-3-5", "reported_at": "2026-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tau3-banking", "languages": [], "modality": "text", "name": "Tau3 Banking", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 33.0, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.33, "raw_min": 0.0567, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau3-banking:grok-4.5", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500"}, "unit": null}, "slug": "llm-stats-tau3-banking", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "τ³-Bench retail domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated online retail environment.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau3-retail:mistral-medium-3-5", "reported_at": "2026-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tau3-retail", "languages": [], "modality": "text", "name": "Tau3 Retail", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.761, "raw_min": 0.761, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau3-retail:mistral-medium-3-5", "reported_date": "2026-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500"}, "unit": null}, "slug": "llm-stats-tau3-retail", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "communication", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "τ³-Bench telecom domain evaluates agentic models on multi-turn, tool-using customer-support and troubleshooting scenarios in a simulated telecommunications environment.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau3-telecom:mistral-medium-3-5", "reported_at": "2026-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tau3-telecom", "languages": [], "modality": "text", "name": "Tau3 Telecom", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.4, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.914, "raw_min": 0.914, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau3-telecom:mistral-medium-3-5", "reported_date": "2026-04-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500"}, "unit": null}, "slug": "llm-stats-tau3-telecom", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TAU3-Bench is a benchmark for evaluating general-purpose agent capabilities, testing models on multi-turn interactions with simulated user models, retrieval, and complex decision-making scenarios.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tau3-bench:qwen3.6-plus", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tau3-bench", "languages": [], "modality": "text", "name": "TAU3-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.89999999999999, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.729, "raw_min": 0.226, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tau3-bench:mimo-v2.5-pro", "reported_date": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-tau3-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TempCompass is a comprehensive benchmark for evaluating temporal perception capabilities of Video Large Language Models (Video LLMs). It constructs conflicting videos that share identical static content but differ in specific temporal aspects to prevent models from exploiting single-frame bias. The benchmark evaluates multiple temporal aspects including action, motion, speed, temporal order, and attribute changes across diverse task formats including multi-choice QA, yes/no QA, caption matching, and caption generation.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tempcompass:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tempcompass", "languages": [], "modality": "multimodal", "name": "TempCompass", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 74.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.748, "raw_min": 0.717, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tempcompass:qwen2.5-vl-72b", "reported_date": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500"}, "unit": null}, "slug": "llm-stats-tempcompass", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Terminal-Bench is a benchmark for testing AI agents in real terminal environments. It evaluates how well agents can handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, security tasks, data science workflows, and cybersecurity vulnerabilities. The benchmark consists of a dataset of ~100 hand-crafted, human-verified tasks and an execution harness that connects language models to a terminal sandbox.", "evidence_summary": {"document_count": 1, "model_count": 25, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:terminal-bench:deepseek-v3.1", "reported_at": "2025-01-10", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:terminal-bench", "languages": [], "modality": "text", "name": "Terminal-Bench", "openness": "restricted", "publisher": null, "released": "2025-01-17", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 25, "score_direction": "higher_is_better", "score_summary": {"display_max": 50.0, "display_multiplier": 100, "model_count": 25, "model_count_basis": "source_model_id", "numeric_count": 25, "raw_max": 0.5, "raw_min": 0.057, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:terminal-bench:claude-sonnet-4-5-20250929", "reported_date": "2025-09-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-terminal-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500", "unit": null}, {"aliases": ["Terminal-Bench", "Terminal Bench", "TerminalBench", "Terminal-Bench 2.0", "Terminal-Bench 2.1", "Terminal Bench 2", "TerminalBench Hard", "Terminal-Bench 3.0", "Terminal Bench 3", "Frontier-Bench", "FrontierBench", "Frontier-Bench v0.1", "FrontierSWE"], "categories": ["agent"], "collected_at": null, "description": "Environment image, timeout and permitted commands change results independently of the model. Version and harness both matter: 2.0 and 2.1 are different instruments, and cards report Terminus, Claude Code and Codex harness numbers for the same version. Frontier-Bench is this series' next version rather than a separate benchmark: frontierbench.ai now bills it as \"Terminal-Bench 3.0 (formerly Frontier-Bench)\", built by the makers of Harbor and Terminal-Bench, so its mentions are counted here. Merging it did not raise this count: all three cards that named Frontier-Bench (Claude Opus 5, Claude Fable 5 and Mythos 5, Kimi K3) also report Terminal-Bench in the same document, and the counting unit is the document, so each still adds one. Being a new version, its task set is not comparable to a 2.x number. Not to be confused with Cognition's FrontierCode, a separate instrument recorded separately despite the shared prefix.", "evidence_summary": {"document_count": 19, "model_count": 13, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000terminal_bench\u0000anthropic_claude_4_system_card\u0000terminal_bench_1\u0000no extended thinking\u0000Claude Opus 4", "reported_at": "2025-05-22", "source_url": "https://www.anthropic.com/news/claude-4"}, "first_score_reported_at": "2025-05-22", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000terminal_bench\u0000anthropic_claude_4_system_card\u0000terminal_bench_1\u0000no extended thinking\u0000Claude Opus 4", "reported_at": "2025-05-22", "source_url": "https://www.anthropic.com/news/claude-4"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:terminal_bench", "languages": [], "modality": null, "name": "Terminal-Bench", "openness": "unknown", "publisher": null, "released": "2025-05-19", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:terminal_bench", "source_url": "https://www.tbench.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 15, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.3, "display_multiplier": 1, "model_count": 13, "model_count_basis": "source_model_id", "numeric_count": 15, "raw_max": 88.3, "raw_min": 43.2, "source_reference": {"obs_id": "curated\u0000terminal_bench\u0000moonshot_kimi_k3_model_card\u0000terminal_bench_2.1\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000terminal_bench\u0000moonshot_kimi_k3_model_card\u0000terminal_bench_2.1\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "terminal_bench", "source": "model_reports", "source_url": "https://www.tbench.ai/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "agents", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Terminal-Bench 2.0 is an updated benchmark for testing AI agents' tool use ability to operate a computer via terminal. It evaluates how well models can handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, security tasks, data science workflows, and cybersecurity vulnerabilities.", "evidence_summary": {"document_count": 1, "model_count": 51, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:terminal-bench-2:qwen3-coder-480b-a35b-instruct", "reported_at": "2025-01-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": true, "key": "llm-stats:terminal-bench-2", "languages": [], "modality": "text", "name": "Terminal-Bench 2.0", "openness": "restricted", "publisher": null, "released": "2025-09-25", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 51, "score_direction": "higher_is_better", "score_summary": {"display_max": 82.69999999999999, "display_multiplier": 100, "model_count": 51, "model_count_basis": "source_model_id", "numeric_count": 51, "raw_max": 0.827, "raw_min": 0.31, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:terminal-bench-2:gpt-5.5", "reported_date": "2026-04-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500"}, "unit": null}, "slug": "llm-stats-terminal-bench-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Terminal-Bench 2.1 is an updated release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal. It evaluates how well models handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, data science workflows, and security tasks.", "evidence_summary": {"document_count": 1, "model_count": 28, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:terminal-bench-2.1:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:terminal-bench-2.1", "languages": [], "modality": "text", "name": "Terminal-Bench 2.1", "openness": "restricted", "publisher": null, "released": "2026-05-05", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 28, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.8, "display_multiplier": 100, "model_count": 28, "model_count_basis": "source_model_id", "numeric_count": 28, "raw_max": 0.888, "raw_min": 0.2458, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:terminal-bench-2.1:gpt-5.6-sol", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500"}, "unit": null}, "slug": "llm-stats-terminal-bench-2-1", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Terminal-Bench 3.0 is a release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal on real-world, end-to-end tasks.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:terminal-bench-3.0:grok-4.6", "reported_at": "2026-08-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:terminal-bench-3.0", "languages": [], "modality": "text", "name": "Terminal-Bench 3.0", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 28.299999999999997, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.283, "raw_min": 0.149, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:terminal-bench-3.0:glm-5.3", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500"}, "unit": null}, "slug": "llm-stats-terminal-bench-3-0", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500", "unit": null}, {"aliases": [], "categories": ["coding", "tool-use"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic coding & terminal use", "evidence_summary": {"document_count": 1, "model_count": 432, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:terminal-bench-hard:9f873c2f-2c2d-4ccb-9e1b-71bf61b052be", "reported_at": "2024-02-26", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-hard"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:terminal-bench-hard", "languages": [], "modality": null, "name": "Terminal-Bench Hard", "openness": "unknown", "publisher": null, "released": "2025-09-02", "released_reference": {"basis": "release_announcement", "note": "Version history dates the introduction of the source-specific Terminal-Bench Hard evaluation to Index v3 on September 2. This is not the release of the broader Terminal-Bench family.", "source_key": "artificial-analysis:terminal-bench-hard", "source_url": "https://artificialanalysis.ai/methodology/intelligence-benchmarking"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 432, "score_direction": "higher_is_better", "score_summary": {"display_max": 65.90909090909089, "display_multiplier": 100, "model_count": 432, "model_count_basis": "source_model_id", "numeric_count": 432, "raw_max": 0.659090909090909, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:terminal-bench-hard:d93edfe8-bf35-49ad-b56e-b18116142a1c", "reported_date": "2026-07-09", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-hard"}, "unit": null}, "slug": "artificial-analysis-terminal-bench-hard", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-hard", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Terminal-Bench Hard is a harder terminal-agent benchmark variant evaluated with the Terminus-2 harness in Cohere's Command A+ and North Mini Code releases.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:terminal-bench-hard:command-a-plus-05-2026", "reported_at": "2026-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:terminal-bench-hard", "languages": [], "modality": "text", "name": "Terminal-Bench Hard", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 31.1, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.311, "raw_min": 0.25, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:terminal-bench-hard:north-mini-code-1.0", "reported_date": "2026-06-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500"}, "unit": null}, "slug": "llm-stats-terminal-bench-hard", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "agentic", "coding", "tool-use"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic coding & terminal use", "evidence_summary": {"document_count": 1, "model_count": 210, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:terminal-bench-v2.1:9f873c2f-2c2d-4ccb-9e1b-71bf61b052be", "reported_at": "2024-02-26", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:terminal-bench-v2.1", "languages": [], "modality": null, "name": "Terminal-Bench v2.1", "openness": "unknown", "publisher": null, "released": "2026-05-06", "released_reference": {"basis": "release_announcement", "note": "The official Terminal-Bench 2.1 release announcement is dated May 6. It precedes adoption in the Artificial Analysis June index update.", "source_key": "artificial-analysis:terminal-bench-v2.1", "source_url": "https://github.com/harbor-framework/terminal-bench-docs/blob/main/content/blog/terminal-bench-2-1.mdx"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 210, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.5131086142322, "display_multiplier": 100, "model_count": 210, "model_count_basis": "source_model_id", "numeric_count": 210, "raw_max": 0.895131086142322, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:terminal-bench-v2.1:d998db47-9b67-4727-a2bb-2e1261020ac0", "reported_date": "2026-07-09", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1"}, "unit": null}, "slug": "artificial-analysis-terminal-bench-v2-1", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Terminal-Bench is a benchmark for testing AI agents in real terminal environments, evaluating how well agents can handle real-world, end-to-end tasks autonomously. The benchmark includes tasks spanning coding, system administration, security, data science, model training, file operations, version control, and web development. Terminus is the neutral test-bed agent designed to work with Terminal-Bench, operating purely through tmux sessions without dedicated tools.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:terminus:kimi-k2-instruct", "reported_at": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:terminus", "languages": [], "modality": "text", "name": "Terminus", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 25.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.25, "raw_min": 0.25, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:terminus:kimi-k2-instruct", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500"}, "unit": null}, "slug": "llm-stats-terminus", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500", "unit": null}, {"aliases": [], "categories": [], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:community:fd462fc2-283c-4967-bd7d-b39d7c661807", "languages": [], "modality": null, "name": "testing", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": "higher_is_better", "score_summary": {"display_max": null, "display_multiplier": 1, "model_count": null, "model_count_basis": null, "numeric_count": 0, "raw_max": null, "raw_min": null, "source_reference": null, "unit": null}, "slug": "llm-stats-community-fd462fc2-283c-4967-bd7d-b39d7c661807", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Afd462fc2-283c-4967-bd7d-b39d7c661807?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "代码工程", "Code", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Text2World, based on planning domain definition language (PDDL), featuring hundreds of diverse domains and employing multi-criteria, execution-based metrics for a more robust evaluation. Text2World 基于规划领域定义语言 (PDDL)，拥有数百个不同的领域，并采用多标准、基于执行的衡量标准来进行更稳健的评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1574", "languages": [], "modality": null, "name": "Text2World", "openness": "unknown", "publisher": "HKU, HIT, Shanghai AI Laboratory, etc.", "released": "2025-02-18", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1574-text2world", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Text2World", "unit": null}, {"aliases": [], "categories": ["multimodal", "image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TextVQA contains 45,336 questions on 28,408 images that require reasoning about text to answer. Introduced to benchmark VQA models' ability to read and reason about text within images, particularly for assistive technologies for visually impaired users. The dataset addresses the gap where existing VQA datasets had few text-based questions or were too small.", "evidence_summary": {"document_count": 1, "model_count": 16, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:textvqa:grok-1.5v", "reported_at": "2024-04-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:textvqa", "languages": [], "modality": "multimodal", "name": "TextVQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 16, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.5, "display_multiplier": 100, "model_count": 16, "model_count_basis": "source_model_id", "numeric_count": 16, "raw_max": 0.855, "raw_min": 0.578, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:textvqa:qwen2-vl-72b", "reported_date": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500"}, "unit": null}, "slug": "llm-stats-textvqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "语言", "Language", "Strong Reasoning", "VQA", "多模态模型", "VLM", "语言理解", "Comprehension", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Temporally-Grounded Language Generation (TGLG) is a benchmark for real-time vision-language models (VLMs) that focus on two key capabilities: perceptual updating and contingency awareness. 基于时间的语言生成（TGLG）是实时视觉语言模型（VLM）的基准，侧重于两个关键功能：感知更新和应急意识。该存储库还包含TGLG的基线实时VLM代码，即具有时间同步交织的视觉语言模型（VLM-TSI）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1846", "languages": [], "modality": "multimodal", "name": "TGLG", "openness": "open", "publisher": "University of Michigan", "released": "2025-05-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1846-tglg", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TGLG", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Thai local dialect benchmark covers Northern (Lanna), Northeastern (Isan), and Southern (Dambro) Thai, evaluating LLMs on five NLP tasks: summarization, question answering, translation, conversation, and food-related tasks. 这是一个涵盖泰国北部（兰纳）、东北部（伊森）和南部（丹布罗）方言的泰国地方方言基准测试，评估大型语言模型在五项自然语言处理任务上的表现：总结、问答、翻译、对话以及与食物相关的任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1734", "languages": ["Thai"], "modality": null, "name": "Thai_local_benchmark", "openness": "unknown", "publisher": "AI Singapore, Vidyasirimedhi Institute of Science and Technology, etc.", "released": "2025-04-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1734-thai-local-benchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Thai_local_benchmark", "unit": null}, {"aliases": [], "categories": ["math", "physics", "reasoning", "finance"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A theorem-driven question answering dataset containing 800 high-quality questions covering 350+ theorems from Math, Physics, EE&CS, and Finance. Designed to evaluate AI models' capabilities to apply theorems to solve challenging university-level science problems.", "evidence_summary": {"document_count": 1, "model_count": 6, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:theoremqa:qwen2-72b-instruct", "reported_at": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:theoremqa", "languages": [], "modality": "text", "name": "TheoremQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 6, "score_direction": "higher_is_better", "score_summary": {"display_max": 44.4, "display_multiplier": 100, "model_count": 6, "model_count_basis": "source_model_id", "numeric_count": 6, "raw_max": 0.444, "raw_min": 0.253, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:theoremqa:qwen2-72b-instruct", "reported_date": "2024-07-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500"}, "unit": null}, "slug": "llm-stats-theoremqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "数理能力", "Math", "逻辑推理", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TheoremQA is the first theorem-driven question-answering dataset designed to evaluate AI models’ capabilities to apply theorems to solve challenging science problems. It is curated by domain experts containing 800 high-quality questions covering 350 theorems from Math, Physics, EE&CS, and Finance. TheoremQA 是第一个基于定理的问题回答数据集，旨在评估 AI 模型应用定理解决复杂科学问题的能力。该数据集由领域专家精心策划，包含 800 个高质量问题，涵盖来自数学、物理、电气与计算机科学以及金融的 350 个定理。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1134", "languages": [], "modality": null, "name": "TheoremQA", "openness": "open", "publisher": "University of Waterloo", "released": "2023-12-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1134-theoremqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TheoremQA", "unit": null}, {"aliases": [], "categories": ["视觉定位", "Visual-Localization", "空间理解", "Spatial-Understanding", "视频理解", "Video-Understanding", "reasoning-VLM", "agentic physical security", "Pulsar VLM", "物理智能", "Embodied AI", "图像理解", "Image Understanding", "Spatial Understanding", "Video Understanding", "逻辑推理", "Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Threat Signature Eval is a domain-specific benchmark for evaluating video-language models on real-world physical security / surveillance footage. It measures how well a model can recognize and classify security-relevant events into a 10-category threat taxonomy (e.g., Fire & Smoke, Fighting & Violen Threat Signature Eval 是一个面向物理安防场景的领域基准，用于评测视频语言模型在真实监控摄像头画面中的理解与事件识别能力。该基准要求模型在安全相关的视频片段中识别并分类事件（10 类威胁签名，如火灾烟雾、斗殴暴力、非法入侵等），相关介绍与结果在 Ambient.ai 的 Pulsar VLM 发布主题演讲中给出。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "opencompass:2391", "languages": [], "modality": "video", "name": "Threat-Signature_Eval", "openness": "unknown", "publisher": "Ambient.ai", "released": "2025-11-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2391-threat-signature-eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Threat-Signature_Eval", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "其他", "Other", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "THUNDER is a benchmark designed to evaluate digital pathology foundation models on tile-level image understanding tasks, facilitating comparative analysis across various downstream tasks and models. THUNDER 是一个用于评估数字病理学基础模型在切片级图像理解任务中表现的基准，旨在支持多种下游任务和模型的对比分析。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2050", "languages": [], "modality": null, "name": "THUNDER", "openness": "unknown", "publisher": "CentraleSupelec , IHU-National PRecISion Medicine Center in Oncology , etc.", "released": "2025-07-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2050-thunder", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/THUNDER", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TimeTravel Taxonomy maps artifacts from 10 civilizations, 266 cultures, and 10k+ verified samples for AI-driven historical analysis. 时间旅行分类将来自 10 个文明、266 个文化以及 10k+个验证样本的文物映射，用于 AI 驱动的历史分析。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1668", "languages": [], "modality": "multimodal", "name": "TimeTravel", "openness": "restricted", "publisher": "Mohamed bin Zayed University of AI, etc.", "released": "2025-02-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1668-timetravel", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TimeTravel", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "Strong Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TinyQA is a benchmark suite designed to evaluate the reasoning abilities of large language models (LLMs). It focuses on assessing LLMs through natural language question-answer pairs, covering various types of reasoning tasks such as causal, logical, and commonsense reasoning. TinyQA是一个用于评估大语言模型（LLMs）推理能力的基准测试套件。该基准专注于通过自然语言问题和答案对来衡量LLMs的推理能力，涵盖了多种类型的推理任务，包括因果推理、逻辑推理和常识推理。TinyQA提供了多样化的数据集，旨在挑战LLMs的推理深度和广度。通过严格的评估，TinyQA能够帮助研究人员更好地理解LLMs在处理复杂语言任务时的表现，并为改进模型提供方向。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1839", "languages": [], "modality": null, "name": "tiny_qa_benchmark_pp", "openness": "unknown", "publisher": "Comet ML", "released": "2025-05-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1839-tiny-qa-benchmark-pp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/tiny_qa_benchmark_pp", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A tool-calling and multimodal interaction benchmark for testing visual instruction following and execution reliability.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tir-bench:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tir-bench", "languages": [], "modality": "multimodal", "name": "TIR-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.6, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.616, "raw_min": 0.532, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tir-bench:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-tir-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["summarization", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A large-scale summarization dataset containing over 9 million training instances extracted from Reddit, designed for extreme summarization (generating one-sentence summaries with high compression and abstraction). More than twice larger than previously proposed datasets.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tldr9+-(test):llama-3.2-3b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tldr9+-(test)", "languages": [], "modality": "text", "name": "TLDR9+ (test)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 19.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.19, "raw_min": 0.19, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tldr9+-(test):llama-3.2-3b-instruct", "reported_date": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500"}, "unit": null}, "slug": "llm-stats-tldr9-test", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TOMATO (Temporal Reasoning Multimodal Evaluation) assesses multimodal models on motion and temporal perception in video, testing understanding of actions, motion, and changes over time.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tomato:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tomato", "languages": [], "modality": "multimodal", "name": "TOMATO", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 79.5, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.795, "raw_min": 0.568, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tomato:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500"}, "unit": null}, "slug": "llm-stats-tomato", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500", "unit": null}, {"aliases": ["Tool Decathlon", "Toolathlon", "Toolathlon-Verified"], "categories": ["tool_use"], "collected_at": null, "description": "Aggregates ten heterogeneous tool suites; the per-suite spread is wide.", "evidence_summary": {"document_count": 8, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000tool_decathlon\u0000zai_glm_5_model_card\u0000tool_decathlon\u0000as reported\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000tool_decathlon\u0000zai_glm_5_model_card\u0000tool_decathlon\u0000as reported\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:tool_decathlon", "languages": [], "modality": null, "name": "Tool Decathlon", "openness": "unknown", "publisher": null, "released": "2025-10-28", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:tool_decathlon", "source_url": "https://toolathlon.xyz/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.5, "display_multiplier": 1, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 76.5, "raw_min": 38.0, "source_reference": {"obs_id": "curated\u0000tool_decathlon\u0000moonshot_kimi_k3_model_card\u0000tool_decathlon\u0000reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000tool_decathlon\u0000moonshot_kimi_k3_model_card\u0000tool_decathlon\u0000reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "tool_decathlon", "source": "model_reports", "source_url": "https://toolathlon.xyz/", "unit": "percent"}, {"aliases": [], "categories": ["reasoning", "agents", "tool_calling"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Tool Decathlon is a comprehensive benchmark for evaluating AI agents' ability to use multiple tools across diverse task categories. It measures proficiency in tool selection, sequencing, and execution across ten different tool-use scenarios.", "evidence_summary": {"document_count": 1, "model_count": 37, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:toolathlon:deepseek-reasoner", "reported_at": "2025-12-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "llm-stats:toolathlon", "languages": [], "modality": "text", "name": "Toolathlon", "openness": "open", "publisher": null, "released": "2025-10-29", "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 37, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.6, "display_multiplier": 100, "model_count": 37, "model_count_basis": "source_model_id", "numeric_count": 37, "raw_max": 0.756, "raw_min": 0.269, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:toolathlon:muse-spark-1.1", "reported_date": "2026-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500"}, "unit": null}, "slug": "llm-stats-toolathlon", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500", "unit": null}, {"aliases": ["Toolathlon Verified", "Toolathlon"], "categories": ["tool_use"], "collected_at": null, "description": "Pass@1 averaged over 3 independent runs. Official evaluation service.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000toolathlon_verified\u0000zai_glm_5_3_flash_model_card\u0000toolathlon_verified\u0000official evaluation service, pass@1 averaged over 3 independent runs\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "first_score_reported_at": "2026-08-27", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000toolathlon_verified\u0000zai_glm_5_3_flash_model_card\u0000toolathlon_verified\u0000official evaluation service, pass@1 averaged over 3 independent runs\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "model-reports:toolathlon_verified", "languages": [], "modality": null, "name": "Toolathlon Verified", "openness": "unknown", "publisher": null, "released": "2025-11-01", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:toolathlon_verified", "source_url": "https://github.com/toolathlon/toolathlon"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.4, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 78.4, "raw_min": 74.1, "source_reference": {"obs_id": "curated\u0000toolathlon_verified\u0000zai_glm_5_3_flash_model_card\u0000toolathlon_verified\u0000official evaluation service, pass@1 averaged over 3 independent runs\u0000GLM-5.3-Flash", "observation_id": "curated\u0000toolathlon_verified\u0000zai_glm_5_3_flash_model_card\u0000toolathlon_verified\u0000official evaluation service, pass@1 averaged over 3 independent runs\u0000GLM-5.3-Flash", "reported_at": "2026-08-27", "reported_date": "2026-08-27", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash"}, "unit": "percent"}, "slug": "toolathlon_verified", "source": "model_reports", "source_url": "https://github.com/toolathlon/toolathlon", "unit": "percent"}, {"aliases": [], "categories": ["智能体", "Agent", "其他", "Other", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ToolHop is a dataset specifically designed for rigorous evaluation of multi-hop tool use, which ensures diverse queries, meaningful interdependencies, locally executable tools, detailed feedback, and verifiable answers through a novel query-driven data construction approach. ToolHop是一个通过查询驱动构建的数据集，专门用于评测大模型的多跳工具使用能力，具备多样化的查询、有意义的相互依赖关系、本地可执行的工具、详细的反馈和可验证的答案五大特征。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1712", "languages": [], "modality": null, "name": "ToolHop", "openness": "unknown", "publisher": "复旦大学&字节跳动", "released": "2025-01-05", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the query-driven multi-hop tool-use benchmark.", "source_key": "opencompass:1712", "source_url": "https://arxiv.org/abs/2501.02506"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1712-toolhop", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ToolHop", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ToolRet is a heterogeneous tool retrieval benchmark comprising 7.6k diverse retrieval tasks, and a corpus of 43k tools, collected from existing datasets. ToolRet是一个包含 7.6k 个不同检索任务的异构工具检索基准，以及从现有数据集中收集的 43k 个工具语料库。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1607", "languages": [], "modality": null, "name": "ToolRet", "openness": "unknown", "publisher": "Shandong University, Baidu Inc, etc.", "released": "2025-03-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1607-toolret", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ToolRet", "unit": null}, {"aliases": [], "categories": ["agents", "coding"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Trae Code Gen is a component of Trae Agent Bench that evaluates implementing new functionality across multiple programming languages in containerized, runnable repositories.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:trae-code-gen:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:trae-code-gen", "languages": [], "modality": "text", "name": "Trae Code Gen", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.624, "raw_min": 0.597, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:trae-code-gen:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500"}, "unit": null}, "slug": "llm-stats-trae-code-gen", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "coding"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Trae Error Fix is a component of Trae Agent Bench that evaluates fixing existing code across multiple programming languages in containerized, runnable repositories.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:trae-error-fix:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:trae-error-fix", "languages": [], "modality": "text", "name": "Trae Error Fix", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 63.3, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.633, "raw_min": 0.567, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:trae-error-fix:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500"}, "unit": null}, "slug": "llm-stats-trae-error-fix", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "合作共建", "Co-Built", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TransBench is the first industry-oriented comprehensive multilingual translation evaluation system designed for industrial applications. It quantifies translation model performance across diverse industries and linguistic environments through meticulously curated datasets aligned with standards. TransBench is the first industry-oriented comprehensive multilingual translation evaluation system designed for industrial applications. It quantifies translation model performance across diverse industries and linguistic environments through meticulously curated datasets aligned with standards.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1855", "languages": ["English", "Chinese", "Japanese", "French", "Arabic", "Multilingual"], "modality": null, "name": "TransBench", "openness": "unknown", "publisher": "Alibaba International Digital Commerce, Beijing Language and Culture University", "released": "2025-05-20", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1855-transbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TransBench", "unit": null}, {"aliases": [], "categories": ["language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "COMET-22 is an ensemble machine translation evaluation metric combining a COMET estimator model trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It demonstrates improved correlations compared to state-of-the-art metrics and increased robustness to critical errors.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:translation-en→set1-comet22:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:translation-en→set1-comet22", "languages": [], "modality": "text", "name": "Translation en→Set1 COMET22", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.1, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.891, "raw_min": 0.885, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:translation-en→set1-comet22:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500"}, "unit": null}, "slug": "llm-stats-translation-en-set1-comet22", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500", "unit": null}, {"aliases": [], "categories": ["language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Translation evaluation using spBLEU (SentencePiece BLEU), a BLEU metric computed over text tokenized with a language-agnostic SentencePiece subword model. Introduced in the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:translation-en→set1-spbleu:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:translation-en→set1-spbleu", "languages": [], "modality": "text", "name": "Translation en→Set1 spBleu", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 43.4, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.434, "raw_min": 0.402, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:translation-en→set1-spbleu:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500"}, "unit": null}, "slug": "llm-stats-translation-en-set1-spbleu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500", "unit": null}, {"aliases": [], "categories": ["language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "COMET-22 is a neural machine translation evaluation metric that uses an ensemble of two models: a COMET estimator trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It provides improved correlations with human judgments and increased robustness to critical errors compared to previous metrics.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:translation-set1→en-comet22:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:translation-set1→en-comet22", "languages": [], "modality": "text", "name": "Translation Set1→en COMET22", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.89, "raw_min": 0.887, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:translation-set1→en-comet22:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500"}, "unit": null}, "slug": "llm-stats-translation-set1-en-comet22", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500", "unit": null}, {"aliases": [], "categories": ["language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "spBLEU (SentencePiece BLEU) evaluation metric for machine translation quality assessment, using language-agnostic SentencePiece tokenization with BLEU scoring. Part of the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:translation-set1→en-spbleu:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:translation-set1→en-spbleu", "languages": [], "modality": "text", "name": "Translation Set1→en spBleu", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 44.4, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.444, "raw_min": 0.426, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:translation-set1→en-spbleu:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500"}, "unit": null}, "slug": "llm-stats-translation-set1-en-spbleu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "智能体", "Agent", "指令跟随", "Instruct", "Legal AI", "任务执行", "Task Execution", "语言理解", "Comprehension", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "本文提出TransLaw——专为香港判例翻译设计的协同交互式多智能体框架。该框架创新性地将传统翻译流程拆解为翻译、错误标注及校对修正三大子任务，并分配三个智能体协同执行。为评估框架性能，我们构建了大规模双语基准数据集BJC Judgments，对13个开源与商业大语言模型（作为智能体）展开评测。实验结果验证了协同策略的有效性：在多智能体协作显著提升效果的同时，提供了具有参考价值的LLM性能横向对比。通过错误类型学分析，本研究进一步揭示了亟待解决的关键翻译挑战。未来工作将聚焦于优化智能体架构以应对这些挑战，同时开发更全面、低成本的评估基准。 本文提出TransLaw——专为香港判例翻译设计的协同交互式多智能体框架。该框架创新性地将传统翻译流程拆解为翻译、错误标注及校对修正三大子任务，并分配三个智能体协同执行。为评估框架性能，我们构建了大规模双语基准数据集BJC Judgments，对13个开源与商业大语言模型（作为智能体）展开评测。实验结果验证了协同策略的有效性：在多智能体协作显著提升效果的同时，提供了具有参考价值的LLM性能横向对比。通过错误类型学分析，本研究进一步揭示了亟待解决的关键翻译挑战。未来工作将聚焦于优化智能体架构以应对这些挑战，同时开发更全面、低成本的评估基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:2039", "languages": [], "modality": null, "name": "TransLaw", "openness": "unknown", "publisher": "City University of Hong Kong", "released": "2025-07-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2039-translaw", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TransLaw", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "spatial_reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TreeBench evaluates visual grounded reasoning, requiring models to localize and reason about fine-grained visual details.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:treebench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:treebench", "languages": [], "modality": "multimodal", "name": "TreeBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.1, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.711, "raw_min": 0.711, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:treebench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500"}, "unit": null}, "slug": "llm-stats-treebench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A large-scale reading comprehension dataset containing over 650K question-answer-evidence triples. TriviaQA includes 95K question-answer pairs authored by trivia enthusiasts and independently gathered evidence documents (six per question on average) that provide high quality distant supervision for answering the questions. The dataset features relatively complex, compositional questions with considerable syntactic and lexical variability, requiring cross-sentence reasoning to find answers.", "evidence_summary": {"document_count": 1, "model_count": 18, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:triviaqa:gemma-2-27b-it", "reported_at": "2024-06-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:triviaqa", "languages": [], "modality": "text", "name": "TriviaQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 18, "score_direction": "higher_is_better", "score_summary": {"display_max": 85.1, "display_multiplier": 100, "model_count": 18, "model_count_basis": "source_model_id", "numeric_count": 18, "raw_max": 0.851, "raw_min": 0.592, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:triviaqa:kimi-k2-base", "reported_date": "2025-07-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500"}, "unit": null}, "slug": "llm-stats-triviaqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TriviaqQA is a reading comprehension dataset containing over 650K question-answer-evidence triples. TriviaqQA includes 95K question-answer pairs authored by trivia enthusiasts and independently gathered evidence documents, six per question on average, that provide high quality distant supervision for answering the questions. TriviaqQA是一个阅读理解数据集，包含超过65万个问题-答案-证据三元组。其包括95K个问答对，由冷知识爱好者和独立收集的事实性文档撰写，平均每个问题6个，为回答问题提供高质量的远程监督。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:512", "languages": [], "modality": null, "name": "TriviaQA", "openness": "unknown", "publisher": null, "released": "2017-05-09", "released_reference": {"basis": "paper_first_version", "note": "First version introducing the TriviaQA question-answering dataset.", "source_key": "opencompass:512", "source_url": "https://arxiv.org/abs/1705.03551"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-512-triviaqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TriviaQA", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning", "finance", "general", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TruthfulQA is a benchmark to measure whether language models are truthful in generating answers to questions. It comprises 817 questions that span 38 categories, including health, law, finance and politics. The questions are crafted such that some humans would answer falsely due to a false belief or misconception, testing models' ability to avoid generating false answers learned from human texts.", "evidence_summary": {"document_count": 1, "model_count": 18, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:truthfulqa:mistral-nemo-instruct-2407", "reported_at": "2024-07-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:truthfulqa", "languages": [], "modality": "text", "name": "TruthfulQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 18, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 18, "model_count_basis": "source_model_id", "numeric_count": 18, "raw_max": 0.88, "raw_min": 0.503, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:truthfulqa:mai-thinking-1", "reported_date": "2026-06-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500"}, "unit": null}, "slug": "llm-stats-truthfulqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "事实可靠性", "Factual Reliability", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TruthfulQA is  a benchmark to measure whether a language model is truthful in generating answers to questions. The benchmark comprises 817 questions that span 38 categories, including health, law, finance and politics. TruthfulQA 用于测量语言模型在回答问题时的真实度。该基准包含 817 个问题，涵盖 38 个类别，包括健康、法律、金融和政治。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1096", "languages": [], "modality": null, "name": "TruthfulQA", "openness": "unknown", "publisher": "OpenAI", "released": "2022-05-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1096-truthfulqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TruthfulQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "TVBench is a temporal video understanding benchmark evaluating reasoning over actions, events, and temporal dynamics in videos.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tvbench:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tvbench", "languages": [], "modality": "multimodal", "name": "TVBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.5, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.805, "raw_min": 0.772, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tvbench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500"}, "unit": null}, "slug": "llm-stats-tvbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multilingual question answering benchmark covering 11 typologically diverse languages with 204K question-answer pairs. Questions are written by people seeking genuine information and data is collected directly in each language without translation to test model generalization across diverse linguistic structures.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:tydiqa:llama-4-maverick", "reported_at": "2025-04-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:tydiqa", "languages": [], "modality": "text", "name": "TydiQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 31.7, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.317, "raw_min": 0.315, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:tydiqa:llama-4-maverick", "reported_date": "2025-04-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500"}, "unit": null}, "slug": "llm-stats-tydiqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "TyDi QA is a question answering dataset covering 11 typologically diverse languages with 204K question-answer pairs. The languages of TyDi QA are diverse with regard to their typology -- the set of linguistic features that each language expresses. TyDi QA 是一个涵盖 11 种不同语言的问题回答数据集，包含 20.4 万个问题-答案对。TyDi QA 的语言种类多样，涵盖了语言学特征的各种类型。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:508", "languages": [], "modality": null, "name": "TyDiQA", "openness": "unknown", "publisher": null, "released": "2020-03-10", "released_reference": {"basis": "paper_first_version", "note": "First version introducing TyDi QA.", "source_key": "opencompass:508", "source_url": "https://arxiv.org/abs/2003.05002"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-508-tydiqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TyDiQA", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "RAG", "Needle In A Haystack", "大语言模型", "LLM", "长上下文", "Long Context", "检索能力", "Retrieval", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "U-NIAH is a framework unifying RAG and LLM for needle-in-a-haystack tasks, based on the fictional Starlight Academy dataset. It eliminates interference from pre-trained knowledge and supports diverse, complex scenarios (e.g., multi-needle, long-needle, \"needle-in-needle\"). U-NIAH是将 RAG 和 LLM 统一映射在大海捞针任务中的框架。所有任务基于一个虚构背景下的数据集Starlight Academy，涵盖了魔法系统、学术课程、校园生活、等多个方面，旨在消除预训练知识的干扰，从而能够独立于 LLMs 的先验知识。框架包含多种评估场景，支持多针（3、7、15个针）和长针（400-500 token）配置，还引入了“针中针”结构，进一步增加了复杂性。该数据集通过多样化的场景和合成生成的内容，能够从多个维度分析模型在长文本场景下的性能。同时通过模块化设计，U-NIAH可持续注入新的挑战场景。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1742", "languages": [], "modality": null, "name": "U-NIAH", "openness": "unknown", "publisher": "同济大学", "released": "2025-03-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1742-u-niah", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/U-NIAH", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "智能体", "Agent", "任务执行", "Task Execution", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "OSWorld is a first-of-its-kind scalable, real computer environment for multimodal agents, supporting task setup, execution-based evaluation, and interactive learning across operating systems. OSWorld 是一个首创的、可扩展的、真实计算机环境，用于多模态智能体，支持操作系统跨平台的任务设置、基于执行的评估和交互式学习。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1625", "languages": [], "modality": "multimodal", "name": "ubuntu_osworld", "openness": "unknown", "publisher": "HKU, etc.", "released": "2024-04-11", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1625-ubuntu-osworld", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ubuntu_osworld", "unit": null}, {"aliases": [], "categories": ["知识", "Knowledge", "大语言模型", "LLM", "知识储备", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "UHGEval(Unconstrained Hallucination Generation Evaluation) benchmark contains hallucinations generated by LLMs with minimal restrictions. UHGEval 基准，包含由限制条件最小的大语言模型（LLMs）生成的幻觉。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1088", "languages": [], "modality": null, "name": "UHGEval", "openness": "open", "publisher": "Institute for Advanced Algorithms Research", "released": "2024-05-24", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1088-uhgeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UHGEval", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "NeurIPS 2024", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "UniBench is meant for evaluating VLMs' reasoning abilities. It is a unified implementation of 50+ VLM benchmarks spanning a comprehensive range of carefully categorized capabilities from object recognition to spatial awareness, counting, and much more. UniBench旨在评估VLM的推理能力，包括50个基准测试，涵盖对象识别、空间感知、计数等任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1332", "languages": ["English"], "modality": "multimodal", "name": "UniBench", "openness": "unknown", "publisher": "Meta FAIR", "released": "2024-08-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1332-unibench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UniBench", "unit": null}, {"aliases": [], "categories": ["legal", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The Uniform Bar Examination (UBE) benchmark evaluates language models on the complete bar exam including multiple-choice Multistate Bar Examination (MBE), open-ended Multistate Essay Exam (MEE), and Multistate Performance Test (MPT) components. Used to assess legal reasoning capabilities across seven subject areas including Evidence, Torts, Constitutional Law, Contracts, Criminal Law and Procedure, Real Property, and Civil Procedure.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:uniform-bar-exam:gpt-4-0613", "reported_at": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:uniform-bar-exam", "languages": [], "modality": "text", "name": "Uniform Bar Exam", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.9, "raw_min": 0.9, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:uniform-bar-exam:gpt-4-0613", "reported_date": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500"}, "unit": null}, "slug": "llm-stats-uniform-bar-exam", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The benchmark is designed to evaluate whether video-large language models (Video-LLMs) can naturally process continuous first-person visual observations like humans, enabling recall, perception, reasoning, and navigation. UrbanVideo-Bench旨在评估视频大型语言模型（Video-LLMs）是否能够像人类一样自然地处理连续的第一人称视觉观察，实现回忆、感知、推理和导航。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1636", "languages": [], "modality": "multimodal", "name": "UrbanVideo-Bench", "openness": "open", "publisher": "THU", "released": "2025-03-08", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1636-urbanvideo-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UrbanVideo-Bench", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "USAMO 2026 evaluates models on the six problems from the 2026 United States of America Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:usamo-2026:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:usamo-2026", "languages": [], "modality": "text", "name": "USAMO 2026", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 36.0, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 36.0, "raw_min": 30.24, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:usamo-2026:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500"}, "unit": null}, "slug": "llm-stats-usamo-2026", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500", "unit": null}, {"aliases": [], "categories": ["math", "reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The 2025 United States of America Mathematical Olympiad (USAMO) benchmark consists of six challenging mathematical problems requiring rigorous proof-based reasoning. USAMO is the most prestigious high school mathematics competition in the United States, serving as the final round of the American Mathematics Competitions series. This benchmark evaluates models on mathematical problem-solving capabilities beyond simple numerical computation, focusing on formal mathematical reasoning and proof generation.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:usamo25:grok-4", "reported_at": "2025-07-09", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:usamo25", "languages": [], "modality": "text", "name": "USAMO25", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.6, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.976, "raw_min": 0.375, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:usamo25:claude-mythos-preview", "reported_date": "2026-04-07", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500"}, "unit": null}, "slug": "llm-stats-usamo25", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "Code Agent", "SWE-Bench", "Automatic test augmentation", "智能体", "Agent", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "UTBoost generates unit tests using LLMs to augment the test cases for certain instances in SWE-Bench, enabling a more rigorous use of SWE-Bench to evaluate the performance of Code Agents. UTBoost通过LLM生成的单元测试，增强了SWE-Bench中一些instances的测试用例，能严谨的使用SWE-Bench来评估Code Agents的表现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2014", "languages": [], "modality": null, "name": "UTBoost", "openness": "unknown", "publisher": null, "released": "2025-06-10", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2014-utboost", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UTBoost", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A visual reasoning benchmark evaluating multimodal inference under challenging spatial and grounded tasks.", "evidence_summary": {"document_count": 1, "model_count": 7, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:v-star:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:v-star", "languages": [], "modality": "multimodal", "name": "V*", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 7, "score_direction": "higher_is_better", "score_summary": {"display_max": 96.89999999999999, "display_multiplier": 100, "model_count": 7, "model_count_basis": "source_model_id", "numeric_count": 7, "raw_max": 0.969, "raw_min": 0.89, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:v-star:kimi-k2.6", "reported_date": "2026-04-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500"}, "unit": null}, "slug": "llm-stats-v-star", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "V-STaR is a spatio-temporal reasoning benchmark for Video-LLMs, evaluating Video-LLM’s spatio-temporal reasoning ability in answering questions explicitly in the context of “when”, “where”, and “what”. V-STaR 是一个针对 Video-LLMs的空间时间推理基准，评估 Video-LLM在“何时”、“何地”和“何物”的上下文中明确回答问题的空间时间推理能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1654", "languages": [], "modality": "multimodal", "name": "V-STaR", "openness": "restricted", "publisher": "CV-Group, Queen Mary University of London, NJU, etc.", "released": "2025-03-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1654-v-star", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/V-STaR", "unit": null}, {"aliases": [], "categories": ["multimodal", "language", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VaTeX: A Large-Scale, High-Quality Multilingual Dataset for Video-and-Language Research. Contains over 41,250 videos and 825,000 captions in both English and Chinese, with over 206,000 English-Chinese parallel translation pairs. Supports multilingual video captioning and video-guided machine translation tasks.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vatex:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vatex", "languages": [], "modality": "multimodal", "name": "VATEX", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.8, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.778, "raw_min": 0.778, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vatex:nova-lite", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500"}, "unit": null}, "slug": "llm-stats-vatex", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "创作", "Creation", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VBench is a comprehensive benchmark evaluates video generation quality. It comprises 16 dimensions in video generation, and also provides a dataset of human preference annotations. VBench用于评估多模态大模型的视频生成质量，包含16个视频生成维度及1个人类偏好注释数据集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1375", "languages": [], "modality": "multimodal", "name": "VBench", "openness": "unknown", "publisher": "Nanyang Technological University", "released": "2023-11-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1375-vbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Visual Commonsense Reasoning (VCR) benchmark that tests higher-order cognition and commonsense reasoning beyond simple object recognition. Models must answer challenging questions about images and provide rationales justifying their answers. The benchmark measures the ability to infer people's actions, goals, and mental states from visual context.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vcr-en-easy:qwen2-vl-72b", "reported_at": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vcr-en-easy", "languages": [], "modality": "multimodal", "name": "VCR_en_easy", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.93, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.9193, "raw_min": 0.9193, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vcr-en-easy:qwen2-vl-72b", "reported_date": "2024-08-29", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500"}, "unit": null}, "slug": "llm-stats-vcr-en-easy", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500", "unit": null}, {"aliases": ["Vending Bench", "Vending Bench 2"], "categories": ["agent"], "collected_at": null, "description": "Reported in simulated dollars, not a percentage, and run over long horizons where a single early failure dominates the total.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000vending_bench\u0000zai_glm_5_model_card\u0000vending_bench_2\u0000runs conducted independently by Andon Labs\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "first_score_reported_at": "2026-02-11", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000vending_bench\u0000zai_glm_5_model_card\u0000vending_bench_2\u0000runs conducted independently by Andon Labs\u0000GLM-5", "reported_at": "2026-02-11", "source_url": "https://huggingface.co/zai-org/GLM-5"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:vending_bench", "languages": [], "modality": null, "name": "Vending Bench", "openness": "unknown", "publisher": null, "released": "2025-02-18", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:vending_bench", "source_url": "https://andonlabs.com/evals/vending-bench-2"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 5634.41, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 5634.41, "raw_min": 4432.12, "source_reference": {"obs_id": "curated\u0000vending_bench\u0000zai_glm_5_1_model_card\u0000vending_bench_2\u0000as reported\u0000GLM-5.1", "observation_id": "curated\u0000vending_bench\u0000zai_glm_5_1_model_card\u0000vending_bench_2\u0000as reported\u0000GLM-5.1", "reported_at": "2026-04-03", "reported_date": "2026-04-03", "source_id": "zai_glm_5_1_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5.1"}, "unit": "usd"}, "slug": "vending_bench", "source": "model_reports", "source_url": "https://andonlabs.com/evals/vending-bench-2", "unit": "usd"}, {"aliases": [], "categories": ["reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Vending-Bench 2 tests longer horizon planning capabilities by evaluating how well AI models can manage a simulated vending machine business over extended periods. The benchmark measures a model's ability to maintain consistent tool usage and decision-making for a full simulated year of operation, driving higher returns without drifting off task.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vending-bench-2:gemini-3-pro-preview", "reported_at": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vending-bench-2", "languages": [], "modality": "text", "name": "Vending-Bench 2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 8017.59, "display_multiplier": 1, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 8017.59, "raw_min": 3635.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vending-bench-2:claude-opus-4-6", "reported_date": "2026-02-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500"}, "unit": null}, "slug": "llm-stats-vending-bench-2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Visual Interface Building Evaluation benchmark for UI/app generation", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe", "languages": [], "modality": "text", "name": "VIBE", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.6, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.886, "raw_min": 0.886, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "SVRPBench is an open and extensible benchmark for the Stochastic Vehicle Routing Problem (SVRP). It includes 500+ instances spanning small to large scales (10–1000 customers), designed to evaluate algorithms under realistic urban logistics conditions with uncertainty and operational constraints. SVRPBench是一个针对随机车辆路径问题（SVRP）的开放且可扩展的基准测试平台。它包含500多个实例，涵盖小到大规模（10-1000个客户），旨在评估算法在具有不确定性和操作约束的现实城市物流条件下的表现。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1875", "languages": [], "modality": null, "name": "VIBE", "openness": "unknown", "publisher": "MBZUAI, Abu Dhabi, UAE", "released": "2025-05-29", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1875-vibe", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VIBE", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE benchmark subset for Android application generation", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-android:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-android", "languages": [], "modality": "text", "name": "VIBE Android", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.897, "raw_min": 0.897, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-android:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-android", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE benchmark subset for backend service generation", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-backend:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-backend", "languages": [], "modality": "text", "name": "VIBE Backend", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.7, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.867, "raw_min": 0.867, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-backend:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-backend", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE benchmark subset for iOS application generation", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-ios:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-ios", "languages": [], "modality": "text", "name": "VIBE iOS", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.88, "raw_min": 0.88, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-ios:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-ios", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE benchmark subset for simulation code generation", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-simulation:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-simulation", "languages": [], "modality": "text", "name": "VIBE Simulation", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.1, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.871, "raw_min": 0.871, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-simulation:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-simulation", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500", "unit": null}, {"aliases": [], "categories": ["code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE benchmark subset for web application generation", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-web:minimax-m2.1", "reported_at": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-web", "languages": [], "modality": "text", "name": "VIBE Web", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 91.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.915, "raw_min": 0.915, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-web:minimax-m2.1", "reported_date": "2025-12-23", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-web", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "general", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE-Eval is a hard evaluation suite for measuring progress of multimodal language models, consisting of 269 visual understanding prompts with gold-standard responses authored by experts. The benchmark has dual objectives: vibe checking multimodal chat models for day-to-day tasks and rigorously testing frontier models, with the hard set containing >50% questions that all frontier models answer incorrectly.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-eval:gemini-1.5-flash-8b", "reported_at": "2024-03-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-eval", "languages": [], "modality": "multimodal", "name": "Vibe-Eval", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.2, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.672, "raw_min": 0.409, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-eval:gemini-2.5-pro-preview-06-05", "reported_date": "2025-06-05", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-eval", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE-Pro is an advanced version of the VIBE (Visual & Interactive Benchmark for Execution) benchmark that evaluates LLMs on professional-grade full-stack application development tasks. It measures model performance across complex real-world development scenarios including web, mobile, and backend applications with higher difficulty than the standard VIBE benchmark.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-pro:minimax-m2.5", "reported_at": "2026-02-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-pro", "languages": [], "modality": "text", "name": "VIBE-Pro", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 55.60000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.556, "raw_min": 0.542, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-pro:minimax-m2.7", "reported_date": "2026-03-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-pro", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VIBE-V2 is an internal benchmark covering pure front-end and full-stack Web, Android, and iOS projects with build-from-scratch tasks. It uses an Agent-as-a-Verifier paradigm to automatically verify program interaction logic and visual output, scoring models through a unified pipeline that includes a requirement set, containerized deployment, and a dynamic interaction environment.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vibe-v2:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vibe-v2", "languages": [], "modality": "text", "name": "VIBE-V2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 50.12, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.5012, "raw_min": 0.5012, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vibe-v2:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500"}, "unit": null}, "slug": "llm-stats-vibe-v2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500", "unit": null}, {"aliases": ["ViBench"], "categories": ["coding_agent"], "collected_at": null, "description": "End-to-end vibe-coding benchmark by authors at Replit and Georgian AI Lab, scoring web applications from the user's perspective. Tasks are derived from anonymized Replit production traces, so the benchmark is first-party to one of the agents it measures. It reaches this registry through a customer testimonial in Anthropic's Claude Fable 5 / Mythos 5 launch post, where it is described qualitatively and never scored in the comparison table. Distinct from VBench, an unrelated video-generation benchmark.", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:vibench", "languages": [], "modality": null, "name": "ViBench", "openness": "unknown", "publisher": null, "released": "2026-05-26", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:vibench", "source_url": "https://vibench.ai/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "vibench", "source": "model_reports", "source_url": "https://vibench.ai/", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Video-MME is the first-ever comprehensive evaluation benchmark of Multi-modal Large Language Models (MLLMs) in video analysis. It features 900 videos totaling 254 hours with 2,700 human-annotated question-answer pairs across 6 primary visual domains (Knowledge, Film & Television, Sports Competition, Life Record, Multilingual, and others) and 30 subfields. The benchmark evaluates models across diverse temporal dimensions (11 seconds to 1 hour), integrates multi-modal inputs including video frames, subtitles, and audio, and uses rigorous manual labeling by expert annotators for precise assessment.", "evidence_summary": {"document_count": 1, "model_count": 17, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:video-mme:gemini-1.5-flash-8b", "reported_at": "2024-03-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:video-mme", "languages": [], "modality": "multimodal", "name": "Video-MME", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 17, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.2, "display_multiplier": 100, "model_count": 17, "model_count_basis": "source_model_id", "numeric_count": 17, "raw_max": 0.892, "raw_min": 0.55, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:video-mme:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500"}, "unit": null}, "slug": "llm-stats-video-mme", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500", "unit": null}, {"aliases": ["Video-MME", "VideoMME"], "categories": ["multimodal"], "collected_at": null, "description": "Frame sampling rate dominates long-video results.", "evidence_summary": {"document_count": 4, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000video_mme\u0000qwen3_5_model_card\u0000video_mme\u0000thinking (temp 0.6, top_p 0.95, top_k 20), with subtitles\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000video_mme\u0000qwen3_5_model_card\u0000video_mme\u0000thinking (temp 0.6, top_p 0.95, top_k 20), with subtitles\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:video_mme", "languages": [], "modality": null, "name": "Video-MME", "openness": "unknown", "publisher": null, "released": "2024-05-31", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:video_mme", "source_url": "https://video-mme.github.io/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.0, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 90.0, "raw_min": 83.7, "source_reference": {"obs_id": "curated\u0000video_mme\u0000moonshot_kimi_k3_model_card\u0000video_mme\u0000with subtitle, reasoning=max\u0000Kimi K3", "observation_id": "curated\u0000video_mme\u0000moonshot_kimi_k3_model_card\u0000video_mme\u0000with subtitle, reasoning=max\u0000Kimi K3", "reported_at": "2026-06-13", "reported_date": "2026-06-13", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3"}, "unit": "percent"}, "slug": "video_mme", "source": "model_reports", "source_url": "https://video-mme.github.io/", "unit": "percent"}, {"aliases": [], "categories": ["多模态", "Multimodal", "理解", "Understanding", "多模态模型", "VLM", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Video-MME is an evaluation benchmark of multi-modal LLMs in video analysis, including 900 videos in various duration with a total of 254 hours which spans 6 primary visual domains with 30 subfields. Video-MME用于评估多模态大模型的视频分析能力，包含900个不同长度的视频，来自6个主要视觉领域和30个子领域，总时长达254小时。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:1358", "languages": [], "modality": "multimodal", "name": "Video-MME", "openness": "unknown", "publisher": "University of Science and Technology of China", "released": "2024-03-31", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1358-video-mme", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Video-MME", "unit": null}, {"aliases": [], "categories": ["multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Video-MME is the first-ever comprehensive evaluation benchmark for Multi-modal Large Language Models (MLLMs) in video analysis. This variant focuses on long-term videos (30min-60min) without subtitle inputs, testing robust contextual dynamics across 6 primary visual domains with 30 subfields including knowledge, film & television, sports competition, life record, and multilingual content.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:video-mme-(long,-no-subtitles):gpt-4.1-2025-04-14", "reported_at": "2025-04-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:video-mme-(long,-no-subtitles)", "languages": [], "modality": "multimodal", "name": "Video-MME (long, no subtitles)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 72.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.72, "raw_min": 0.72, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:video-mme-(long,-no-subtitles):gpt-4.1-2025-04-14", "reported_date": "2025-04-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500"}, "unit": null}, "slug": "llm-stats-video-mme-long-no-subtitles", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["创作", "Creation", "NeurIPS 2024", "智能体", "Agent", "任务执行", "Task Execution", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VideoGUI is designed to evaluate GUI assistants on visual-centric GUI tasks. Sourced from high-quality web instructional videos, it focuses on tasks involving professional and novel software and complex activities (e.g., video editing). VideoGUI旨在评估以视觉为中心的GUI任务上的GUI助手，来自高质量的网络教学视频，侧重于涉及专业和新颖软件和复杂活动（例如视频编辑）的任务。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1277", "languages": [], "modality": null, "name": "VideoGUI", "openness": "unknown", "publisher": "National University of Singapore", "released": "2024-06-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1277-videogui", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VideoGUI", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "video"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VideoHolmes evaluates video understanding and reasoning capabilities in multimodal models.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:videoholmes:mimo-v2.5", "reported_at": "2026-04-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:videoholmes", "languages": [], "modality": "multimodal", "name": "VideoHolmes", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.2, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.682, "raw_min": 0.64, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:videoholmes:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500"}, "unit": null}, "slug": "llm-stats-videoholmes", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "逻辑推理", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VideoMathQA is a benchmark designed to evaluate mathematical reasoning in real-world educational videos. It requires models to interpret and integrate information from three modalities, visuals, audio, and text, across time. VideoMathQA是一个旨在评估实际教育视频中数学推理能力的基准。它要求模型解释和整合来自三种模态(视觉、音频和文本)随时间变化的信息。该基准解决了\"多模态针堆\"问题,即关键信息稀疏且分散在视频的不同模态和时刻。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1926", "languages": [], "modality": "multimodal", "name": "VideoMathQA", "openness": "unknown", "publisher": "MBZUAI,University of California Merced,Google Research,etc", "released": "2025-06-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1926-videomathqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VideoMathQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The first-ever comprehensive evaluation benchmark of Multi-modal LLMs in Video analysis. Features 900 videos (254 hours) with 2,700 question-answer pairs covering 6 primary visual domains and 30 subfields. Evaluates temporal understanding across short (11 seconds) to long (1 hour) videos with multi-modal inputs including video frames, subtitles, and audio.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:videomme-w-sub.:qwen2.5-vl-7b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:videomme-w-sub.", "languages": [], "modality": "multimodal", "name": "VideoMME w sub.", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.4, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.904, "raw_min": 0.716, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:videomme-w-sub.:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500"}, "unit": null}, "slug": "llm-stats-videomme-w-sub", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Video-MME is a comprehensive evaluation benchmark for multi-modal large language models in video analysis. It features 900 videos across 6 primary visual domains with 30 subfields, ranging from 11 seconds to 1 hour in duration, with 2,700 question-answer pairs. The benchmark evaluates MLLMs' capabilities in processing sequential visual data and multi-modal content including video frames, subtitles, and audio.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:videomme-w-o-sub.:qwen2.5-vl-72b", "reported_at": "2025-01-26", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:videomme-w-o-sub.", "languages": [], "modality": "multimodal", "name": "VideoMME w/o sub.", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.89999999999999, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.839, "raw_min": 0.651, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:videomme-w-o-sub.:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500"}, "unit": null}, "slug": "llm-stats-videomme-w-o-sub", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Video-MMMU evaluates Large Multimodal Models' ability to acquire knowledge from expert-level professional videos across six disciplines through three cognitive stages: perception, comprehension, and adaptation. Contains 300 videos and 900 human-annotated questions spanning Art, Business, Science, Medicine, Humanities, and Engineering.", "evidence_summary": {"document_count": 1, "model_count": 26, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:videommmu:gpt-4o-2024-08-06", "reported_at": "2024-08-06", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "llm-stats:videommmu", "languages": [], "modality": "multimodal", "name": "VideoMMMU", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "benchmark_source", "repo_resolution_status": "resolved", "score_count": 26, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.6, "display_multiplier": 100, "model_count": 26, "model_count_basis": "source_model_id", "numeric_count": 26, "raw_max": 0.876, "raw_min": 0.562, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:videommmu:gemini-3-pro-preview", "reported_date": "2025-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500"}, "unit": null}, "slug": "llm-stats-videommmu", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "多模态", "Multimodal", "video reasoning", "MLLMs", "多模态模型", "VLM", "逻辑推理", "Reasoning", "视频理解", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VideoReasonBench is designed to evaluate vision-centric complex video reasoning. VideoReasonBench是一个用于评测视觉为中心、复杂视频推理的基准。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1900", "languages": [], "modality": "multimodal", "name": "VideoReasonBench", "openness": "restricted", "publisher": "Peking University", "released": "2025-06-05", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1900-videoreasonbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VideoReasonBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "knowledge", "video", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VideoSimpleQA evaluates factual knowledge grounded in video content, measuring how accurately models answer short, fact-seeking questions about videos.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:videosimpleqa:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:videosimpleqa", "languages": [], "modality": "multimodal", "name": "VideoSimpleQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 76.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.764, "raw_min": 0.714, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:videosimpleqa:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500"}, "unit": null}, "slug": "llm-stats-videosimpleqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "vision-language", "math", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ViLBench is a benchmark designed to evaluate vision-language models. It features 600 examples from 5 datasets, selected based on the criterion that process reward models offer greater improvements over output reward models in guiding generations. ViLBench 是一项旨在评估视觉-语言模型的数据集，其强调对模型进行细粒度的逐步推理能力测试。该基准共包含600个经过严格筛选的样本，来源于五个不同的视觉-语言数据集，筛选标准是在模型答案选择过程中，过程奖励模型（process-reward model）相较于输出奖励模型（output-reward model）具有更显著的性能提升效果。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1717", "languages": [], "modality": "multimodal", "name": "ViLBench", "openness": "unknown", "publisher": "UC Santa Cruz", "released": "2025-03-26", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1717-vilbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ViLBench", "unit": null}, {"aliases": [], "categories": ["safety", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Virology Capabilities Test (VCT) is an expert-level multiple-choice benchmark measuring the capability to troubleshoot complex virology laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vct:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vct", "languages": [], "modality": "text", "name": "Virology Capabilities Test", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.61, "raw_min": 0.61, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vct:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500"}, "unit": null}, "slug": "llm-stats-vct", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "知识", "Knowledge", "self-critique", "VLM", "VLM reasoning", "多模态模型", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VISCO aims to evaluate the critique and correction capabilities of VLMs, which are two essential building blocks towards VLM self-improvement. VISCO requires VLMs to critique the correctness of each step in CoT, provide natural language explanation, and corrects the CoT based on the critique. VISCO 旨在评估 VLM 的评判（critique）和纠正（correction）能力，这两个能力是 VLM 自主提升推理性能的基础。对于一个视觉推理问题，给定模型生成的 CoT，VISCO 评测集要求 VLM 评判 CoT 中每个步骤的正确性，提供自然语言解释，并根据评判结果对 CoT 进行修正。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2063", "languages": [], "modality": "multimodal", "name": "VISCO", "openness": "unknown", "publisher": "UCLA", "released": "2024-12-03", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2063-visco", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VISCO", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VisFactor is a benchmark evaluating fine-grained visual factor perception and reasoning over images.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:visfactor:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:visfactor", "languages": [], "modality": "multimodal", "name": "VisFactor", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.4, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.514, "raw_min": 0.428, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:visfactor:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500"}, "unit": null}, "slug": "llm-stats-visfactor", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "code", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Vision2Web evaluates multimodal models on converting visual designs and screenshots into functional web pages, measuring end-to-end design-to-code capability.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vision2web:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vision2web", "languages": [], "modality": "multimodal", "name": "Vision2Web", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 69.0, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.69, "raw_min": 0.31, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vision2web:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500"}, "unit": null}, "slug": "llm-stats-vision2web", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "创作", "Creation", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ViStoryBench is a benchmark designed to rigorously evaluate the capabilities of multimodal generative models (e.g., diffusion models, LLM-based agents) in synthesizing visually coherent image sequences from textual narratives and reference images. ViStoryBench 是一个面向故事可视化任务的综合性评测基准，旨在评估多模态生成模型（如扩散模型视频生成模型等）根据给定叙事文本和参考图像生成视觉连贯且情节一致的图像序列的能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1993", "languages": ["English", "Chinese"], "modality": "multimodal", "name": "ViStoryBench", "openness": "open", "publisher": "Shanghai Tech University , StepFun , AIGC Research , AGI Lab, Westlake Universit", "released": "2025-06-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1993-vistorybench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ViStoryBench", "unit": null}, {"aliases": [], "categories": ["强推理", "Strong Reasoning", "多模态", "Multimodal", "推理", "Reasoning", "多模态模型", "VLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VisualPuzzles is a benchmark that targets visual reasoning while deliberately minimizing reliance on specialized knowledge. VisualPuzzles consists of 1168 diverse questions spanning five categories: algorithmic, analogical, deductive, inductive, and spatial reasoning. LLM 能考公务员吗？我们做了个测试…\n\n近年来，大模型（LLM）的能力突飞猛进，似乎“越来越聪明”了。但有一个关键问题仍然摆在眼前：\n🤔 它们真的会“推理”吗？\n\n🚀 我们设计了一个名为 VisualPuzzles 🧩 的全新数据集，专门用来回答这个问题：\n脱离专业知识的支持，大模型能靠逻辑本身解题吗？\n我们从多个来源精心挑选或改编了 1168 道图文逻辑题，其中一个重要来源便是中国国家公务员考试行测中的逻辑推理题（没错，真·考公难度）🎯", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1748", "languages": [], "modality": "multimodal", "name": "VisualPuzzles", "openness": "unknown", "publisher": "Carnegie Mellon University", "released": "2025-04-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1748-visualpuzzles", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VisualPuzzles", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "事实可靠性", "Factual Reliability", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VisualSimpleQA is a multimodal fact-seeking benchmark with two key features. VisualSimpleQA 是一个多模态事实寻求基准，具有两个关键特性。首先，它使视觉和语言模态中 LVLMs 的评估更加简化和解耦。其次，它纳入了明确的难度标准，以指导人工标注并促进提取具有挑战性的子集，即 VisualSimpleQA-hard。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": true, "key": "opencompass:1634", "languages": [], "modality": "multimodal", "name": "VisualSimpleQA", "openness": "restricted", "publisher": "Zhongguancun Laboratory, RUC, Tencent, etc.", "released": "2025-03-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1634-visualsimpleqa", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VisualSimpleQA", "unit": null}, {"aliases": [], "categories": ["multimodal", "frontend_development", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A multimodal benchmark designed to assess the capabilities of multimodal large language models (MLLMs) across web page understanding and grounding tasks. Comprises 7 tasks (captioning, webpage QA, heading OCR, element OCR, element grounding, action prediction, and action grounding) with 1.5K human-curated instances from 139 real websites across 87 sub-domains.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:visualwebbench:nova-lite", "reported_at": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:visualwebbench", "languages": [], "modality": "multimodal", "name": "VisualWebBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 79.7, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.797, "raw_min": 0.777, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:visualwebbench:nova-pro", "reported_date": "2024-11-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500"}, "unit": null}, "slug": "llm-stats-visualwebbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VisuLogic evaluates logical reasoning capabilities in visual contexts.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:visulogic:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:visulogic", "languages": [], "modality": "multimodal", "name": "VisuLogic", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.543, "display_multiplier": 1, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.543, "raw_min": 0.344, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:visulogic:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500"}, "unit": null}, "slug": "llm-stats-visulogic", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VITA-Bench evaluates AI agents on real-world virtual task automation, measuring their ability to complete complex multi-step tasks in simulated environments.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vita-bench:qwen3.5-397b-a17b", "reported_at": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vita-bench", "languages": [], "modality": "text", "name": "VITA-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 49.7, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.497, "raw_min": 0.22, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vita-bench:qwen3.5-397b-a17b", "reported_date": "2026-02-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-vita-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "视频理解", "Video-Understanding", "Visual Knowledge", "Physics", "Psychology", "物理智能", "Embodied AI", "逻辑推理", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "尽管多模态大型语言模型（MLLMs）在识别物体方面已相当熟练，但它们往往缺乏对世界潜在物理和社会原则的直觉的、类似人类的理解。这种高级的、基于视觉的语义，我们称之为视觉知识，其构成了感知与推理之间的桥梁。为了系统地评估这种能力，我们提出了 VKnowU，一个包含 1,249 个视频、1,680 个问题的综合基准，涵盖了 8 种核心类型的视觉知识，既包括以世界为中心的（例如直觉物理），也包括以人类为中心的任务（例如主观意图）。 尽管多模态大型语言模型（MLLMs）在识别物体方面已相当熟练，但它们往往缺乏对世界潜在物理和社会原则的直觉的、类似人类的理解。这种高级的、基于视觉的语义，我们称之为视觉知识，其构成了感知与推理之间的桥梁。为了系统地评估这种能力，我们提出了 VKnowU，一个包含 1,249 个视频、1,680 个问题的综合基准，涵盖了 8 种核心类型的视觉知识，既包括以世界为中心的（例如直觉物理），也包括以人类为中心的任务（例如主观意图）。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:2388", "languages": [], "modality": "multimodal", "name": "VKnowU", "openness": "restricted", "publisher": "Shanghai AI Laboratory", "released": "2025-11-25", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2388-vknowu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VKnowU", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VLADBench is a vision-language autonomous-driving benchmark evaluating understanding of dynamic traffic scenes and participants.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vladbench:qwen3.7-plus", "reported_at": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vladbench", "languages": [], "modality": "multimodal", "name": "VLADBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 77.2, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.772, "raw_min": 0.772, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vladbench:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500"}, "unit": null}, "slug": "llm-stats-vladbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VLM²-Bench is the first comprehensive benchmark that evaluates vision-language models' (VLMs) ability to visually link matching cues across multi-image sequences and videos. The benchmark consists of 9 subtasks with over 3,000 test cases. VLM²-Bench 是第一个全面评估视觉语言模型（VLMs）在多图像序列和视频中视觉链接匹配线索能力的基准。该基准包括 9 个子任务，超过 3000 个测试案例，旨在评估人类日常使用的根本视觉链接能力。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1541", "languages": [], "modality": "multimodal", "name": "VLM2-Bench", "openness": "restricted", "publisher": "HKUST; CMU; MIT", "released": "2025-02-17", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1541-vlm2-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VLM2-Bench", "unit": null}, {"aliases": [], "categories": ["multimodal", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VLMsAreBiased evaluates whether vision-language models rely on visual evidence or fall back on language priors when answering.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vlmsarebiased:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vlmsarebiased", "languages": [], "modality": "multimodal", "name": "VLMsAreBiased", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.6, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.836, "raw_min": 0.683, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vlmsarebiased:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500"}, "unit": null}, "slug": "llm-stats-vlmsarebiased", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A vision-language benchmark that probes blind spots and brittle reasoning in multimodal models.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vlmsareblind:qwen3.5-122b-a10b", "reported_at": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vlmsareblind", "languages": [], "modality": "multimodal", "name": "VLMsAreBlind", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.0, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.97, "raw_min": 0.967, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vlmsareblind:qwen3.5-35b-a3b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500"}, "unit": null}, "slug": "llm-stats-vlmsareblind", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500", "unit": null}, {"aliases": [], "categories": ["audio"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A dataset for improving human vocal sounds recognition, containing over 21,000 crowdsourced recordings of laughter, sighs, coughs, throat clearing, sneezes, and sniffs from 3,365 unique subjects. Used for audio event classification and recognition of human non-speech vocalizations.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vocalsound:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vocalsound", "languages": [], "modality": "audio", "name": "VocalSound", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 93.89999999999999, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.939, "raw_min": 0.939, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vocalsound:qwen2.5-omni-7b", "reported_date": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500"}, "unit": null}, "slug": "llm-stats-vocalsound", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "safety", "speech_to_text", "general", "communication"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VoiceBench is the first benchmark designed to provide a multi-faceted evaluation of LLM-based voice assistants, evaluating capabilities including general knowledge, instruction-following, reasoning, and safety using both synthetic and real spoken instruction data with diverse speaker characteristics and environmental conditions.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:voicebench-avg:qwen2.5-omni-7b", "reported_at": "2025-03-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:voicebench-avg", "languages": [], "modality": "multimodal", "name": "VoiceBench Avg", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 90.10000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.901, "raw_min": 0.7412, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:voicebench-avg:inkling-small", "reported_date": "2026-07-30", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500"}, "unit": null}, "slug": "llm-stats-voicebench-avg", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "image_to_text", "healthcare", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VQA-RAD (Visual Question Answering in Radiology) is the first manually constructed dataset of medical visual question answering containing 3,515 clinically generated visual questions and answers about radiology images. The dataset includes questions created by clinical trainees on 315 radiology images from MedPix covering head, chest, and abdominal scans, designed to support AI development for medical image analysis and improve patient care.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vqa-rad:medgemma-4b-it", "reported_at": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vqa-rad", "languages": [], "modality": "multimodal", "name": "VQA-Rad", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 49.9, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.499, "raw_min": 0.499, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vqa-rad:medgemma-4b-it", "reported_date": "2025-05-20", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500"}, "unit": null}, "slug": "llm-stats-vqa-rad", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VQAv2 is a balanced Visual Question Answering dataset that addresses language bias by providing complementary images for each question, forcing models to rely on visual understanding rather than language priors. It contains approximately twice the number of image-question pairs compared to the original VQA dataset.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vqav2:pixtral-12b-2409", "reported_at": "2024-09-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vqav2", "languages": [], "modality": "multimodal", "name": "VQAv2", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 80.9, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.809, "raw_min": 0.781, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vqav2:pixtral-large", "reported_date": "2024-11-18", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500"}, "unit": null}, "slug": "llm-stats-vqav2", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "image_to_text", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VQA v2.0 (Visual Question Answering v2.0) is a balanced dataset designed to counter language priors in visual question answering. It consists of complementary image pairs where the same question yields different answers, forcing models to rely on visual understanding rather than language bias. The dataset contains 1,105,904 questions across 204,721 COCO images, requiring understanding of vision, language, and commonsense knowledge.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vqav2-(test):llama-3.2-11b-instruct", "reported_at": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vqav2-(test)", "languages": [], "modality": "multimodal", "name": "VQAv2 (test)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.2, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.752, "raw_min": 0.752, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vqav2-(test):llama-3.2-11b-instruct", "reported_date": "2024-09-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500"}, "unit": null}, "slug": "llm-stats-vqav2-test", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "image_to_text", "language", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "VQAv2 is a balanced Visual Question Answering dataset containing open-ended questions about images that require understanding of vision, language, and commonsense knowledge to answer. VQAv2 addresses bias issues from the original VQA dataset by collecting complementary images such that every question is associated with similar images that result in different answers, forcing models to actually understand visual content rather than relying on language priors.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:vqav2-(val):gemma-3-12b-it", "reported_at": "2025-03-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:vqav2-(val)", "languages": [], "modality": "multimodal", "name": "VQAv2 (val)", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 71.6, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.716, "raw_min": 0.624, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:vqav2-(val):gemma-3-12b-it", "reported_date": "2025-03-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500"}, "unit": null}, "slug": "llm-stats-vqav2-val", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "视频理解", "Video-Understanding", "多步推理", "视频推理", "多语种", "多模态模型", "VLM", "逻辑推理", "Video Understanding", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "VRBench is the first evaluation benchmark specifically designed to assess multi-step reasoning capabilities over long videos. The benchmark comprises 1,010 long videos with an average duration of 1.6 hours, covering 8 languages and 7 video types. It includes annotations for 9,468 reasoning steps. VRBench是首个专门针对长视频多步推理能力设计的评测基准VRBench。该基准包含 1010 个平均时长1.6小时的长视频，覆盖 8 种语言和 7 种视频类型，标注了 9468 个多步推理问答对和超过 3 万个详细推理步骤。本评测集能够同时对模型的推理过程和推理结果进行评分，是首个同时具备多步推理链标注和评测能力的视频评测集。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:2383", "languages": [], "modality": "multimodal", "name": "VRBench", "openness": "open", "publisher": "上海人工智能实验室", "released": "2025-06-12", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2383-vrbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VRBench", "unit": null}, {"aliases": [], "categories": ["math", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "We-Math evaluates multimodal models on visual mathematical reasoning, requiring models to understand and solve math problems presented with visual elements such as diagrams, charts, and geometric figures.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:we-math:qwen3.6-plus", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:we-math", "languages": [], "modality": "multimodal", "name": "We-Math", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 89.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.89, "raw_min": 0.89, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:we-math:qwen3.6-plus", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500"}, "unit": null}, "slug": "llm-stats-we-math", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500", "unit": null}, {"aliases": [], "categories": ["agents", "coding"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Web Bench evaluates agents on realistic web-development engineering tasks, measuring end-to-end implementation in browser-based workflows.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:web-bench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:web-bench", "languages": [], "modality": "text", "name": "Web Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 78.4, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.784, "raw_min": 0.736, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:web-bench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-web-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WebArena-Verified evaluates browser agents on realistic web tasks using a verified task set and execution-based grading.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:webarena-verified:qwen3.8-27b", "reported_at": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:webarena-verified", "languages": [], "modality": "multimodal", "name": "WebArena-Verified", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.8, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.648, "raw_min": 0.648, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:webarena-verified:qwen3.8-27b", "reported_date": "2026-08-14", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500"}, "unit": null}, "slug": "llm-stats-webarena-verified", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "frontend_development", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WebDev Arena is a leaderboard for evaluating AI models on web development tasks, including zero-shot generation, complex prompts, and interactive web UI creation. Models are ranked using Elo ratings based on their performance in coding and web development challenges.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:webdev-arena:gemini-3.7-flash", "reported_at": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:webdev-arena", "languages": [], "modality": "text", "name": "WebDev Arena", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 1588.0, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 1588.0, "raw_min": 1588.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:webdev-arena:gemini-3.7-flash", "reported_date": "2026-08-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500"}, "unit": null}, "slug": "llm-stats-webdev-arena", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "智能体", "Agent", "任务执行", "Task Execution", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WebUIBench is a comprehensive benchmark designed to evaluate Multimodal Large Language Models (MLLMs) in Web UI-to-code generation tasks across four key capabilities: UI perception, HTML programming, UI-code understanding, and end-to-end transformation. WebUIBench 是一个面向多模态大语言模型（MLLMs）的综合性评测基准，旨在系统评估模型在 Web UI 到代码生成任务中的四个关键能力：界面感知、HTML 编程、界面-代码理解以及整体转换能力。该基准包含来自 700 多个真实网站的 21,000 个高质量问答对，支持对模型在各阶段的细粒度能力分析。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1982", "languages": [], "modality": null, "name": "WebUI-Bench", "openness": "open", "publisher": "The Chinese University of HongKong,HongKong SAR,China ,etc.", "released": "2025-06-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1982-webui-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WebUI-Bench", "unit": null}, {"aliases": [], "categories": ["agents", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WebVoyager evaluates an agent's ability to navigate and complete tasks on real websites by perceiving page screenshots and executing browser actions.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:webvoyager:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:webvoyager", "languages": [], "modality": "multimodal", "name": "WebVoyager", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.5, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.885, "raw_min": 0.885, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:webvoyager:glm-5v-turbo", "reported_date": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500"}, "unit": null}, "slug": "llm-stats-webvoyager", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "NeurIPS 2024", "任务执行", "Task Execution", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WhodunitBench is used to evaluate large multimodal agent under complex tasks and dynamic scenarios. WhodunitBench用于评估大型多模式代理在复杂任务场景下的动态评估。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": true, "has_size": false, "key": "opencompass:1279", "languages": [], "modality": null, "name": "WhodunitBench", "openness": "unknown", "publisher": "The Chinese University of Hong Kong", "released": "2024-09-26", "released_reference": {"basis": "paper_first_version", "note": "OpenReview's public publication date for the murder-mystery benchmark, before the later camera-ready modification.", "source_key": "opencompass:1279", "source_url": "https://openreview.net/forum?id=qmvtDIfbmS"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1279-whodunitbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WhodunitBench", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WiC is a benchmark for the evaluation of context-sensitive word embeddings. WiC is framed as a binary classification task. Each instance in WiC has a target word w, either a verb or a noun, for which two contexts are provided. Each of these contexts triggers a specific meaning of w. The task is to identify if the occurrences of w in the two contexts correspond to the same meaning or not. In fact, the dataset can also be viewed as an application of Word Sense Disambiguation in practise. Word-in-Context是一个词义消歧任务，被视为句子对的二元分类。给定两个文本片段和一个在两个句子中都出现的多义词，任务是确定该词在两个句子中是否具有相同的含义。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:504", "languages": [], "modality": null, "name": "WiC", "openness": "unknown", "publisher": null, "released": "2018-08-28", "released_reference": {"basis": "paper_first_version", "note": "First version introducing Word-in-Context.", "source_key": "opencompass:504", "source_url": "https://arxiv.org/abs/1808.09121"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-504-wic", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WiC", "unit": null}, {"aliases": [], "categories": ["reasoning", "search", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WideSearch is an agentic search benchmark that evaluates models' ability to perform broad, parallel search operations across multiple sources. It tests wide-coverage information retrieval and synthesis capabilities.", "evidence_summary": {"document_count": 1, "model_count": 10, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:widesearch:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:widesearch", "languages": [], "modality": "text", "name": "WideSearch", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 10, "score_direction": "higher_is_better", "score_summary": {"display_max": 81.89999999999999, "display_multiplier": 100, "model_count": 10, "model_count_basis": "source_model_id", "numeric_count": 10, "raw_max": 0.819, "raw_min": 0.571, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:widesearch:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500"}, "unit": null}, "slug": "llm-stats-widesearch", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500", "unit": null}, {"aliases": ["WideSearch"], "categories": ["agent"], "collected_at": null, "description": "Wide-coverage web collection; scored on completeness of an enumerated answer set.", "evidence_summary": {"document_count": 2, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000widesearch\u0000qwen3_5_model_card\u0000widesearch\u0000thinking (temp 0.6, top_p 0.95, top_k 20), 256k context, no context management\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "first_score_reported_at": "2026-02-16", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000widesearch\u0000qwen3_5_model_card\u0000widesearch\u0000thinking (temp 0.6, top_p 0.95, top_k 20), 256k context, no context management\u0000Qwen3.5-397B-A17B", "reported_at": "2026-02-16", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "model-reports:widesearch", "languages": [], "modality": null, "name": "WideSearch", "openness": "unknown", "publisher": null, "released": "2025-08-11", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:widesearch", "source_url": "https://widesearch-seed.github.io/"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 83.9, "display_multiplier": 1, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 83.9, "raw_min": 74.0, "source_reference": {"obs_id": "curated\u0000widesearch\u0000tencent_hy4_preview\u0000widesearch\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000widesearch\u0000tencent_hy4_preview\u0000widesearch\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "widesearch", "source": "model_reports", "source_url": "https://widesearch-seed.github.io/", "unit": "percent"}, {"aliases": [], "categories": ["推理", "Reasoning", "NeurIPS 2024", "大语言模型", "LLM", "事实可靠性", "Factual Reliability", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WikiContradict is a benchmark consisting of 253 high-quality, human-annotated instances designed to assess LLM performance when augmented with retrieved passages containing real-world knowledge conflicts. WikiContradict旨在评估LLM遇到包含真实世界知识冲突的段落检索增强时的性能，由253个高质量的人工注释实例组成。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1318", "languages": ["English"], "modality": null, "name": "WikiContradict", "openness": "unknown", "publisher": "IBM Research", "released": "2024-06-19", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1318-wikicontradict", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WikiContradict", "unit": null}, {"aliases": [], "categories": ["其他", "Other", "大语言模型", "LLM", "代码工程", "Code", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WikiSQL is a dataset of 80654 hand-annotated examples of questions and SQL queries distributed across 24241 tables from Wikipedia that is an order of magnitude larger than comparable datasets. WikiSQL 是一个包含 80,654 个手动标注示例的问题和 SQL 查询的数据集，分布在来自维基百科的 24,241 个表格中。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1129", "languages": [], "modality": null, "name": "WikiSQL", "openness": "unknown", "publisher": "Salesforce Research", "released": "2017-11-09", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1129-wikisql", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WikiSQL", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "communication"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WildBench is an automated evaluation framework that benchmarks large language models using 1,024 challenging, real-world tasks selected from over one million human-chatbot conversation logs. It introduces two evaluation metrics (WB-Reward and WB-Score) that achieve high correlation with human preferences and uses task-specific checklists for systematic evaluation.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:wild-bench:jamba-1.5-large", "reported_at": "2024-08-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:wild-bench", "languages": [], "modality": "text", "name": "Wild Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 68.5, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.685, "raw_min": 0.424, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:wild-bench:ministral-3-14b-instruct-2512", "reported_date": "2025-12-04", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-wild-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言生成", "Generation", "指令遵循", "Instruction Following", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "Weintroduce WildBench, an automated evaluation framework designed to bench-mark large language models (LLMs) using challenging, real-world user queries, which consists of 1,024 tasks carefully selected from over one million human-chatbot conversation logs. WildBench推出自动评估框架和数据集，基于真实用户难题评测大语言模型，包含从逾百万人机对话日志中精选的1,024个任务样本。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1555", "languages": [], "modality": null, "name": "WildBench", "openness": "restricted", "publisher": "Allen Institute for AI, University of Washington", "released": "2024-06-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1555-wildbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WildBench", "unit": null}, {"aliases": [], "categories": ["agents", "coding"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WildClawBench is an agentic coding benchmark from InternLM/Claw-Eval that reports overall model performance on real-world tool-using development tasks.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:wildclawbench:mimo-v2.5-pro", "reported_at": "2026-04-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:wildclawbench", "languages": [], "modality": "text", "name": "WildClawBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 62.8, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.628, "raw_min": 0.43, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:wildclawbench:seed-2.1-turbo", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500"}, "unit": null}, "slug": "llm-stats-wildclawbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["智能体", "Agent", "代码", "Code", "创作", "Creation", "Agent Evaluation", "In-the-Wild Tasks", "Multimodal & Tool-Use Reasoning", "代码工程", "语言生成", "Generation", "任务执行", "Task Execution", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WildClawBench evaluates AI agents across six task categories: productivity flow, code intelligence, social interaction, search & retrieval, creative synthesis, and safety alignment. It comprises 60 hand-authored tasks inside a live environment with real tools — browser, bash, file system, etc. WildClawBench 从六个任务维度评测 AI 智能体：生产力流程、代码智能、社交互动、搜索与检索、创意合成和安全对齐。基准包含 60 个人工设计的任务，运行在配备真实工具（浏览器、bash、文件系统、电子邮件等）的真实环境中。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": false, "has_repo": true, "has_size": true, "key": "opencompass:2445", "languages": ["English", "Chinese"], "modality": null, "name": "WildClawBench", "openness": "open", "publisher": "上海人工智能实验室", "released": "2026-04-07", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-2445-wildclawbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WildClawBench", "unit": null}, {"aliases": [], "categories": ["reasoning", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WinoGrande: An Adversarial Winograd Schema Challenge at Scale. A large-scale dataset of 44,000 pronoun resolution problems designed to test machine commonsense reasoning. Uses adversarial filtering to reduce spurious biases and provides a more robust evaluation of whether AI systems truly understand commonsense or exploit statistical shortcuts. Current best AI methods achieve 59.4-79.1% accuracy, significantly below human performance of 94.0%.", "evidence_summary": {"document_count": 1, "model_count": 22, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:winogrande:gpt-4-0613", "reported_at": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "llm-stats:winogrande", "languages": [], "modality": "text", "name": "Winogrande", "openness": "unknown", "publisher": "Allen Institute for Artificial Intelligence", "released": "2019-11-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 22, "score_direction": "higher_is_better", "score_summary": {"display_max": 87.5, "display_multiplier": 100, "model_count": 22, "model_count_basis": "source_model_id", "numeric_count": 22, "raw_max": 0.875, "raw_min": 0.513, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:winogrande:gpt-4-0613", "reported_date": "2023-06-13", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500"}, "unit": null}, "slug": "llm-stats-winogrande", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "开源收录", "Open-Source", "支持", "Supported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WINOGRANDE is a large-scale dataset of 44k problems, inspired by the original WSC design, but adjusted to improve both the scale and the hardness of the dataset. WINOGRANDE 包含 44,000 个问题，受到 WSC 设计的启发，但进行了调整，以提高数据集的规模和难度。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1109", "languages": [], "modality": null, "name": "WinoGrande", "openness": "unknown", "publisher": "Allen Institute for Artificial Intelligence", "released": "2019-11-21", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1109-winogrande", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WinoGrande", "unit": null}, {"aliases": [], "categories": ["safety", "healthcare", "biology", "chemistry"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Weapons of Mass Destruction (WMDP) is a multiple-choice benchmark on dual-use biology, chemistry, and cyber knowledge. It measures a model's capacity to enable malicious actors to design, synthesize, acquire, or use chemical, biological, radiological, or nuclear (CBRN) weapons.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:wmdp:grok-4.1-thinking-2025-11-17", "reported_at": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:wmdp", "languages": [], "modality": "text", "name": "WMDP", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 84.0, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.84, "raw_min": 0.84, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:wmdp:grok-4.1-thinking-2025-11-17", "reported_date": "2025-11-17", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500"}, "unit": null}, "slug": "llm-stats-wmdp", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500", "unit": null}, {"aliases": [], "categories": ["language", "healthcare"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "The Eighth Conference on Machine Translation (WMT23) benchmark evaluating machine translation systems across 8 language pairs (14 translation directions) including general, biomedical, literary, and low-resource language translation tasks. Features specialized shared tasks for quality estimation, metrics evaluation, sign language translation, and discourse-level literary translation with professional human assessment.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:wmt23:gemini-1.0-pro", "reported_at": "2024-02-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:wmt23", "languages": [], "modality": "text", "name": "WMT23", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 75.1, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.751, "raw_min": 0.717, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:wmt23:gemini-1.5-pro", "reported_date": "2024-05-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500"}, "unit": null}, "slug": "llm-stats-wmt23", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500", "unit": null}, {"aliases": [], "categories": ["language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WMT24++ is a comprehensive multilingual machine translation benchmark that expands the WMT24 dataset to cover 55 languages and dialects. It includes human-written references and post-edits across four domains (literary, news, social, and speech) to evaluate machine translation systems and large language models across diverse linguistic contexts.", "evidence_summary": {"document_count": 1, "model_count": 23, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:wmt24++:gemma-3-12b-it", "reported_at": "2025-03-12", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:wmt24++", "languages": [], "modality": "text", "name": "WMT24++", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": "not_found", "repo_resolution_status": "not_found", "score_count": 23, "score_direction": "higher_is_better", "score_summary": {"display_max": 86.67, "display_multiplier": 100, "model_count": 23, "model_count_basis": "source_model_id", "numeric_count": 23, "raw_max": 0.8667, "raw_min": 0.272, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:wmt24++:nemotron-3-super-120b-a12b", "reported_date": "2026-03-11", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500"}, "unit": null}, "slug": "llm-stats-wmt24", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Workspace Bench evaluates AI agents on high-economic-value workplace tasks that span multi-step planning, file processing, and tool use across realistic office and productivity workflows.", "evidence_summary": {"document_count": 1, "model_count": 3, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:workspace-bench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:workspace-bench", "languages": [], "modality": "text", "name": "Workspace Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 3, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.7, "display_multiplier": 100, "model_count": 3, "model_count_basis": "source_model_id", "numeric_count": 3, "raw_max": 0.677, "raw_min": 0.53, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:workspace-bench:qwen3.8-max", "reported_date": "2026-08-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-workspace-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500", "unit": null}, {"aliases": ["Workspace-Bench", "Workspace Bench", "Workspace-Bench 1.0"], "categories": ["professional"], "collected_at": null, "description": "Large-scale file-workspace tasks; the environment and file dependencies are part of the measurement.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "document_publication", "obs_id": "curated\u0000workspacebench\u0000tencent_hy4_preview\u0000workspacebench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "first_score_reported_at": "2026-08-28", "first_score_source_reference": {"date_precision": "document_publication", "obs_id": "curated\u0000workspacebench\u0000tencent_hy4_preview\u0000workspacebench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "has_dataset": false, "has_paper": true, "has_repo": false, "has_size": false, "key": "model-reports:workspacebench", "languages": [], "modality": null, "name": "WorkspaceBench", "openness": "unknown", "publisher": null, "released": "2026-05-05", "released_reference": {"basis": "benchmark_release", "source_key": "model-reports:workspacebench", "source_url": "https://arxiv.org/abs/2605.03596"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 60.2, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 60.2, "raw_min": 60.2, "source_reference": {"obs_id": "curated\u0000workspacebench\u0000tencent_hy4_preview\u0000workspacebench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "observation_id": "curated\u0000workspacebench\u0000tencent_hy4_preview\u0000workspacebench\u0000as printed in the report's benchmark table\u0000Hy4 preview", "reported_at": "2026-08-28", "reported_date": "2026-08-28", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview"}, "unit": "percent"}, "slug": "workspacebench", "source": "model_reports", "source_url": "https://arxiv.org/abs/2605.03596", "unit": "percent"}, {"aliases": [], "categories": ["multimodal", "knowledge", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WorldBench evaluates real-world visual knowledge and understanding across diverse everyday scenes.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:worldbench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:worldbench", "languages": [], "modality": "multimodal", "name": "WorldBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 67.60000000000001, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.676, "raw_min": 0.637, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:worldbench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500"}, "unit": null}, "slug": "llm-stats-worldbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "推理", "Reasoning", "知识", "Knowledge", "多模态模型", "VLM", "视觉生成", "Visual Generation", "知识储备", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": false, "has_size": false, "key": "opencompass:1938", "languages": [], "modality": "multimodal", "name": "WorldGenBench", "openness": "unknown", "publisher": null, "released": "2025-06-16", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1938-worldgenbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WorldGenBench", "unit": null}, {"aliases": [], "categories": ["创作", "Creation", "其他", "Other", "视频生成", "多模态模型", "VLM", "视觉生成", "Visual Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WorldScore benchmark is the first unified benchmark for world generation. WorldScore基准测试，这是首个用于世界生成的统一基准测试。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1735", "languages": [], "modality": null, "name": "WorldScore", "openness": "open", "publisher": "Stanford University", "released": "2025-04-01", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1735-worldscore", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WorldScore", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "WorldVQA is a benchmark designed to evaluate atomic vision-centric world knowledge. It assesses models' ability to understand and reason about visual elements representing real-world knowledge.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:worldvqa:kimi-k2.5", "reported_at": "2026-01-27", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:worldvqa", "languages": [], "modality": "multimodal", "name": "WorldVQA", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.1, "display_multiplier": 100, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.611, "raw_min": 0.463, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:worldvqa:qwen3.7-plus", "reported_date": "2026-05-31", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500"}, "unit": null}, "slug": "llm-stats-worldvqa", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500", "unit": null}, {"aliases": [], "categories": ["legal", "finance", "communication", "creativity", "writing"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "A comprehensive benchmark for evaluating large language models' generative writing capabilities across 6 core writing domains (Academic & Engineering, Finance & Business, Politics & Law, Literature & Art, Education, Advertising & Marketing) and 100 subdomains. Contains 1,239 queries with a query-dependent evaluation framework that dynamically generates 5 instance-specific assessment criteria for each writing task, using a fine-tuned critic model to score responses on style, format, and length dimensions.", "evidence_summary": {"document_count": 1, "model_count": 15, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:writingbench:qwen3-235b-a22b-instruct-2507", "reported_at": "2025-07-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:writingbench", "languages": [], "modality": "text", "name": "WritingBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 15, "score_direction": "higher_is_better", "score_summary": {"display_max": 88.3, "display_multiplier": 100, "model_count": 15, "model_count_basis": "source_model_id", "numeric_count": 15, "raw_max": 0.883, "raw_min": 0.738, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:writingbench:qwen3-235b-a22b-thinking-2507", "reported_date": "2025-07-25", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500"}, "unit": null}, "slug": "llm-stats-writingbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["长文本", "Long-Context", "创作", "Creation", "大语言模型", "LLM", "长上下文", "Long Context", "语言生成", "Generation", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WritingBench: A Comprehensive Benchmark for Generative Writing WritingBench: A Comprehensive Benchmark for Generative Writing", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1701", "languages": [], "modality": null, "name": "WritingBench", "openness": "unknown", "publisher": "Alibaba Group; Renmin University of China; Shanghai Jiao Tong University", "released": "2025-03-07", "released_reference": {"basis": "paper_first_version", "note": "First version introducing WritingBench.", "source_key": "opencompass:1701", "source_url": "https://arxiv.org/abs/2503.05244"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1701-writingbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WritingBench", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "WSC is a pronoun disambiguation task, which requires to determine which noun the pronoun refers to according to the context. WSC是一个代词消歧任务，要求根据上下文判断代词指代的是哪个名词。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": "2019-05-02", "first_score_source_reference": {"basis": "score_publication", "date_precision": "score_publication", "note": "The original benchmark release day is absent from the crawl. Table 2 provides the earliest dated numeric language-model evaluation retained in this date archive for the source-linked WSC task; it is not a benchmark release date.", "reported_at": "2019-05-02", "score_evidence": {"locator": "Table 2, BERT row, WSC column", "metric": "accuracy", "model": "BERT-large-cased (fine-tuned)", "unit": "percent", "value": 68.5}, "source_key": "opencompass:507", "source_url": "https://arxiv.org/abs/1905.00537v1"}, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "opencompass:507", "languages": [], "modality": null, "name": "WSC", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-507-wsc", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WSC", "unit": null}, {"aliases": [], "categories": ["代码", "Code", "ACL 2024", "大语言模型", "LLM", "代码工程", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "xCodeEval is the largest executable multilingual multitask benchmark to date consisting of 25M document-level coding examples (16.5B tokens) from about 7.5K unique problems covering up to 11 programming languages with execution-level parallelism. xCodeEval 是迄今为止最大的可执行多语言多任务基准，包含 2500 万个文档级编码示例（165 亿个标记），来自约 7500 个独特问题，涵盖多达 11 种编程语言。它包括 7 个任务，涉及代码理解、生成、翻译和检索。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1072", "languages": ["Multilingual"], "modality": null, "name": "xCodeEval", "openness": "open", "publisher": "NTU-NLP", "released": "2023-11-06", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1072-xcodeeval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/xCodeEval", "unit": null}, {"aliases": [], "categories": ["reasoning", "general", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "xDailyBench evaluates AI agents on white-collar office work, covering everyday professional tasks such as document handling, consultation, and multi-step productivity workflows.", "evidence_summary": {"document_count": 1, "model_count": 2, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:xdailybench:seed-2.1-pro", "reported_at": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:xdailybench", "languages": [], "modality": "text", "name": "xDailyBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 2, "score_direction": "higher_is_better", "score_summary": {"display_max": 61.0, "display_multiplier": 100, "model_count": 2, "model_count_basis": "source_model_id", "numeric_count": 2, "raw_max": 0.61, "raw_min": 0.564, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:xdailybench:seed-2.1-pro", "reported_date": "2026-06-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500"}, "unit": null}, "slug": "llm-stats-xdailybench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500", "unit": null}, {"aliases": [], "categories": ["summarization", "language"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "Large-scale multilingual abstractive summarization dataset comprising 1 million professionally annotated article-summary pairs from BBC, covering 44 languages. XL-Sum is highly abstractive, concise, and of high quality, designed to encourage research on multilingual abstractive summarization tasks.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:xlsum-english:llama-3.1-nemotron-70b-instruct", "reported_at": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:xlsum-english", "languages": [], "modality": "text", "name": "XLSum English", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 31.61, "display_multiplier": 100, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 0.3161, "raw_min": 0.3161, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:xlsum-english:llama-3.1-nemotron-70b-instruct", "reported_date": "2024-10-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500"}, "unit": null}, "slug": "llm-stats-xlsum-english", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500", "unit": null}, {"aliases": [], "categories": ["safety"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "XSTest is a test suite designed to identify exaggerated safety behaviours in large language models. It comprises 450 prompts: 250 safe prompts across ten prompt types that well-calibrated models should not refuse to comply with, and 200 unsafe prompts as contrasts that models should refuse. The benchmark systematically evaluates whether models refuse to respond to clearly safe prompts due to overly cautious safety mechanisms.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:xstest:gemini-1.5-flash-8b", "reported_at": "2024-03-15", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:xstest", "languages": [], "modality": "text", "name": "XSTest", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 98.8, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.988, "raw_min": 0.926, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:xstest:gemini-1.5-pro", "reported_date": "2024-05-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500"}, "unit": null}, "slug": "llm-stats-xstest", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500", "unit": null}, {"aliases": [], "categories": ["理解", "Understanding", "大语言模型", "LLM", "语言理解", "Comprehension", "开源收录", "Open-Source", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "XSum is a single-document summarization task which does not favor extractive strategies and calls for an abstractive modeling approach. The idea is to create a short, one-sentence news summary answering the question “What is the article about?”. The dataset collects real-world, large scale data by harvesting online articles from the British Broadcasting Corporation (BBC). XSum是一个单文档摘要任务，不支持抽取式策略，需要采用抽象建模方法。其思想是创建一个简短的一句话新闻摘要，回答“这篇文章是关于什么的？”的问题。该数据集通过从英国广播公司（BBC）收集在线文章，得到了大量的现实数据。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:521", "languages": [], "modality": null, "name": "XSum", "openness": "unknown", "publisher": null, "released": "2018-08-27", "released_reference": {"basis": "paper_first_version", "note": "First version of the paper introducing the Extreme Summarization dataset.", "source_key": "opencompass:521", "source_url": "https://arxiv.org/abs/1808.08745"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-521-xsum", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/XSum", "unit": null}, {"aliases": [], "categories": ["推理", "Reasoning", "大语言模型", "LLM", "逻辑推理", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "xVerify, an efficient answer verifier for reasoning model evaluations. xVerify demonstrates strong capability in equivalence judgment, enabling it to effectively determine whether the answers produced by reasoning models are equivalent to reference answers across various types of objective questions xVerify，这是一种用于推理模型评估的高效答案验证器。xVerify 在等价判断方面表现出强大的能力，使其能够有效地确定推理模型生成的答案是否等同于各种类型客观问题的参考答案。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1784", "languages": ["English", "Chinese"], "modality": null, "name": "xVerify", "openness": "unknown", "publisher": "Peking University，Research Institute of China Telecom，etc.", "released": "2025-04-14", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1784-xverify", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/xVerify", "unit": null}, {"aliases": [], "categories": ["finance", "agents"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "YC-Bench evaluates agents on long-horizon, open-ended business and investment decision-making. The reported metric is the final assets (fund value, in US dollars) accumulated by the agent over the course of the simulation.", "evidence_summary": {"document_count": 1, "model_count": 1, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:yc-bench:minimax-m3", "reported_at": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:yc-bench", "languages": [], "modality": "text", "name": "YC-Bench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 1, "score_direction": "higher_is_better", "score_summary": {"display_max": 2100000.0, "display_multiplier": 1, "model_count": 1, "model_count_basis": "source_model_id", "numeric_count": 1, "raw_max": 2100000.0, "raw_min": 2100000.0, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:yc-bench:minimax-m3", "reported_date": "2026-06-01", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500"}, "unit": null}, "slug": "llm-stats-yc-bench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500", "unit": null}, {"aliases": [], "categories": ["语言", "Language", "大语言模型", "LLM", "语言理解", "Comprehension", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "The benchmarks introduced for evaluating large language models (LLMs) on Cantonese include Yue-TruthfulQA, Yue-GSM8K, Yue-ARC-C, Yue-MMLU, and Yue-TRANS. Each of these benchmarks focuses on different aspects of language understanding and generation in Cantonese, offering a comprehensive means of assessing the capabilities of LLMs in handling this language. Yue_Benchmark用于评估粤语大型语言模型（LLMs）。该评测集包含：Yue TruthtyQA、Yue-GSM8K、Yue-ARC-C、Yue MMLU和Yue TRANS，侧重于粤语语言理解和生成的不同方面，为评估LLM粤语能力提供了一种全面的方法。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": true, "key": "opencompass:1017", "languages": ["English"], "modality": null, "name": "Yue_Benchmark", "openness": "open", "publisher": null, "released": "2024-08-31", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1017-yue-benchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Yue_Benchmark", "unit": null}, {"aliases": [], "categories": ["agents", "code"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ZClawBench evaluates Claw-style agent task execution quality, measuring a model's ability to autonomously complete complex multi-step coding tasks in real-world environments.", "evidence_summary": {"document_count": 1, "model_count": 4, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:zclawbench:glm-5v-turbo", "reported_at": "2026-04-02", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:zclawbench", "languages": [], "modality": "text", "name": "ZClawBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 4, "score_direction": "higher_is_better", "score_summary": {"display_max": 64.3, "display_multiplier": 100, "model_count": 4, "model_count_basis": "source_model_id", "numeric_count": 4, "raw_max": 0.643, "raw_min": 0.526, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:zclawbench:qwen3.7-max", "reported_date": "2026-05-19", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500"}, "unit": null}, "slug": "llm-stats-zclawbench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500", "unit": null}, {"aliases": [], "categories": ["reasoning"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ZebraLogic is an evaluation framework for assessing large language models' logical reasoning capabilities through logic grid puzzles derived from constraint satisfaction problems (CSPs). The benchmark consists of 1,000 programmatically generated puzzles with controllable and quantifiable complexity, revealing a 'curse of complexity' where model accuracy declines significantly as problem complexity grows.", "evidence_summary": {"document_count": 1, "model_count": 8, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:zebralogic:minimax-m1-40k", "reported_at": "2025-06-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:zebralogic", "languages": [], "modality": "text", "name": "ZebraLogic", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 8, "score_direction": "higher_is_better", "score_summary": {"display_max": 97.3, "display_multiplier": 100, "model_count": 8, "model_count_basis": "source_model_id", "numeric_count": 8, "raw_max": 0.973, "raw_min": 0.801, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:zebralogic:qwen3-vl-235b-a22b-thinking", "reported_date": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500"}, "unit": null}, "slug": "llm-stats-zebralogic", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ZEROBench is a challenging vision benchmark designed to test models on zero-shot visual understanding tasks.", "evidence_summary": {"document_count": 1, "model_count": 9, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:zerobench:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:zerobench", "languages": [], "modality": "image", "name": "ZEROBench", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 9, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.41, "display_multiplier": 1, "model_count": 9, "model_count_basis": "source_model_id", "numeric_count": 9, "raw_max": 0.41, "raw_min": 0.04, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:zerobench:kimi-k3", "reported_date": "2026-07-16", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500"}, "unit": null}, "slug": "llm-stats-zerobench", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500", "unit": null}, {"aliases": [], "categories": ["多模态", "Multimodal", "多模态模型", "VLM", "跨模态推理", "Cross-modal Reasoning", "不支持", "Unsupported"], "collected_at": "2026-08-18T00:00:00+00:00", "description": "ZeroBench is a challenging visual reasoning benchmark for LMMs. It consists of a main set of 100 high-quality, manually curated questions covering numerous domains, reasoning types and image type. Questions have been designed and calibrated to be beyond the capabilities of current frontier models. ZeroBench 是针对多模态模型（LMMs）的具有挑战性的视觉推理基准。它由一组主要的 100 个高质量人工问题组成，涵盖多个领域、推理类型和图像类型。ZeroBench 中的问题经过设计和校准，已经超出了当前前沿模型的能力范围。", "evidence_summary": {"document_count": 1, "model_count": null, "model_count_basis": null}, "first_score_record": null, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": true, "has_paper": true, "has_repo": true, "has_size": false, "key": "opencompass:1532", "languages": [], "modality": "multimodal", "name": "ZeroBench", "openness": "restricted", "publisher": "University of Cambridge, University of Alberta,etc.", "released": "2025-02-13", "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 0, "score_direction": null, "score_summary": null, "slug": "opencompass-1532-zerobench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ZeroBench", "unit": null}, {"aliases": [], "categories": ["multimodal", "reasoning", "vision"], "collected_at": "2026-08-17T18:49:35.489925+00:00", "description": "ZEROBench-Sub is a subset of the ZEROBench benchmark.", "evidence_summary": {"document_count": 1, "model_count": 5, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "llm_stats:zerobench-sub:qwen3-vl-235b-a22b-thinking", "reported_at": "2025-09-22", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "llm-stats:zerobench-sub", "languages": [], "modality": "image", "name": "ZEROBench-Sub", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 5, "score_direction": "higher_is_better", "score_summary": {"display_max": 0.362, "display_multiplier": 1, "model_count": 5, "model_count_basis": "source_model_id", "numeric_count": 5, "raw_max": 0.362, "raw_min": 0.277, "source_reference": {"crawled_at": "2026-08-17T18:49:35.489925+00:00", "obs_id": "llm_stats:zerobench-sub:qwen3.5-122b-a10b", "reported_date": "2026-02-24", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500"}, "unit": null}, "slug": "llm-stats-zerobench-sub", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500", "unit": null}, {"aliases": [], "categories": ["agentic", "tool-use"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic tool use", "evidence_summary": {"document_count": 1, "model_count": 440, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:tau2-bench-telecom:217b34ec-5920-4fc1-8886-6a70a324837d", "reported_at": "2023-09-27", "source_url": "https://artificialanalysis.ai/evaluations/tau2-bench"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:tau2-bench-telecom", "languages": [], "modality": null, "name": "τ²-Bench Telecom", "openness": "unknown", "publisher": null, "released": "2025-06-12", "released_reference": {"basis": "release_announcement", "note": "Version 0.1.0 is labelled the initial public release and explicitly includes the Telecom domain. This is not the older tau-bench airline or retail release.", "source_key": "artificial-analysis:tau2-bench-telecom", "source_url": "https://github.com/sierra-research/tau2-bench/blob/main/RELEASE_NOTES.md"}, "repo_kind": null, "repo_resolution_status": null, "score_count": 440, "score_direction": "higher_is_better", "score_summary": {"display_max": 99.1228070175439, "display_multiplier": 100, "model_count": 440, "model_count_basis": "source_model_id", "numeric_count": 440, "raw_max": 0.991228070175439, "raw_min": 0.0, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:tau2-bench-telecom:0121d27b-5b8a-4901-8684-80589cc6d40d", "reported_date": "2026-05-14", "source_url": "https://artificialanalysis.ai/evaluations/tau2-bench"}, "unit": null}, "slug": "artificial-analysis-tau2-bench-telecom", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/tau2-bench", "unit": null}, {"aliases": [], "categories": ["intelligence-index", "agentic", "tool-use"], "collected_at": "2026-08-25T10:41:06Z", "description": "Agentic tool use", "evidence_summary": {"document_count": 1, "model_count": 178, "model_count_basis": "source_model_id"}, "first_score_record": {"date_precision": "model_announcement", "obs_id": "artificial_analysis:tau3-banking:b5c1c91a-7474-4409-9a9c-9c2ac45d9eb6", "reported_at": "2024-07-18", "source_url": "https://artificialanalysis.ai/evaluations/tau3-banking"}, "first_score_reported_at": null, "first_score_source_reference": null, "has_dataset": false, "has_paper": false, "has_repo": false, "has_size": false, "key": "artificial-analysis:tau3-banking", "languages": [], "modality": null, "name": "τ³-Banking", "openness": "unknown", "publisher": null, "released": null, "released_reference": null, "repo_kind": null, "repo_resolution_status": null, "score_count": 178, "score_direction": "higher_is_better", "score_summary": {"display_max": 51.340206185567006, "display_multiplier": 100, "model_count": 178, "model_count_basis": "source_model_id", "numeric_count": 178, "raw_max": 0.51340206185567, "raw_min": 0.00206185567010309, "source_reference": {"crawled_at": "2026-08-25T10:41:06Z", "obs_id": "artificial_analysis:tau3-banking:5e5b4ce7-bc54-47b2-b911-21b9cad8394c", "reported_date": "2026-08-03", "source_url": "https://artificialanalysis.ai/evaluations/tau3-banking"}, "unit": null}, "slug": "artificial-analysis-tau3-banking", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/tau3-banking", "unit": null}], "count": 1284, "document_registry": {"benchmark_count": 1284, "document_count": 1209, "documents": [{"benchmarks": [{"benchmark_id": "artificial-analysis-aa-analystagent", "domain": "agentic", "name": "AA-AnalystAgent", "released": null, "url": "https://artificialanalysis.ai/evaluations/aa-analyst-agent"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/aa-analyst-agent", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/aa-analyst-agent", "title": "AA-AnalystAgent"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-aa-briefcase", "domain": "agentic", "name": "AA-Briefcase", "released": "2026-06-18", "url": "https://artificialanalysis.ai/evaluations/aa-briefcase"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/aa-briefcase", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/aa-briefcase", "title": "AA-Briefcase"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-aime-2025", "domain": "reasoning", "name": "AIME 2025", "released": "2025-02-12", "url": "https://artificialanalysis.ai/evaluations/aime-2025"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/aime-2025", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/aime-2025", "title": "AIME 2025"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-apex-agents-aa", "domain": "agentic", "name": "APEX-Agents-AA", "released": null, "url": "https://artificialanalysis.ai/evaluations/apex-agents-aa"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/apex-agents-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/apex-agents-aa", "title": "APEX-Agents-AA"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-aa-lcr", "domain": "intelligence-index", "name": "AA-LCR", "released": "2025-08-05", "url": "https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning", "title": "AA-LCR"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-automationbench-aa", "domain": "agentic", "name": "AutomationBench-AA", "released": null, "url": "https://artificialanalysis.ai/evaluations/automationbench-aa"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/automationbench-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/automationbench-aa", "title": "AutomationBench-AA"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-critpt", "domain": "intelligence-index", "name": "CritPt", "released": "2025-11-21", "url": "https://artificialanalysis.ai/evaluations/critpt"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/critpt", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/critpt", "title": "CritPt"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-enterpriseops-gym-aa", "domain": "agentic", "name": "EnterpriseOps-Gym-AA", "released": null, "url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa", "title": "EnterpriseOps-Gym-AA"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-gdpval-aa-v2-normalized-score", "domain": "intelligence-index", "name": "GDPval-AA v2 (normalized score)", "released": "2026-06-15", "url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}, {"benchmark_id": "artificial-analysis-gdpval-aa-v2-raw-elo", "domain": "intelligence-index", "name": "GDPval-AA v2 (Elo)", "released": "2026-06-15", "url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/gdpval-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/gdpval-aa", "title": "GDPval-AA v2 (normalized score)"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-gpqa-diamond", "domain": "intelligence-index", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://artificialanalysis.ai/evaluations/gpqa-diamond"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/gpqa-diamond", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/gpqa-diamond", "title": "GPQA Diamond"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-harvey-lab-aa", "domain": "agentic", "name": "Harvey LAB-AA", "released": null, "url": "https://artificialanalysis.ai/evaluations/harvey-lab-aa"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/harvey-lab-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/harvey-lab-aa", "title": "Harvey LAB-AA"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-humanitys-last-exam", "domain": "intelligence-index", "name": "Humanity’s Last Exam", "released": "2025-01-23", "url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/humanitys-last-exam", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam", "title": "Humanity’s Last Exam"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-ifbench", "domain": "instruction-following", "name": "IFBench", "released": "2025-07-03", "url": "https://artificialanalysis.ai/evaluations/ifbench"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/ifbench", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/ifbench", "title": "IFBench"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-itbench-aa", "domain": "agentic", "name": "ITBench-AA", "released": null, "url": "https://artificialanalysis.ai/evaluations/itbench-aa"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/itbench-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/itbench-aa", "title": "ITBench-AA"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://artificialanalysis.ai/evaluations/livecodebench"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/livecodebench", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/livecodebench", "title": "LiveCodeBench"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-mlcr-aa", "domain": "reasoning", "name": "MLCR-AA", "released": null, "url": "https://artificialanalysis.ai/evaluations/mlcr-aa"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/mlcr-aa", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/mlcr-aa", "title": "MLCR-AA"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-mmmu-pro", "domain": "multimodal", "name": "MMMU-Pro", "released": "2024-09-05", "url": "https://artificialanalysis.ai/evaluations/mmmu-pro"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/mmmu-pro", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/mmmu-pro", "title": "MMMU-Pro"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-aa-omniscience-accuracy", "domain": "intelligence-index", "name": "AA-Omniscience Accuracy", "released": "2025-11-16", "url": "https://artificialanalysis.ai/evaluations/omniscience"}, {"benchmark_id": "artificial-analysis-aa-omniscience-non-hallucination", "domain": "intelligence-index", "name": "AA-Omniscience Non-Hallucination Rate", "released": "2025-11-16", "url": "https://artificialanalysis.ai/evaluations/omniscience"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/omniscience", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/omniscience", "title": "AA-Omniscience Accuracy"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-scicode", "domain": "intelligence-index", "name": "SciCode", "released": "2024-07-18", "url": "https://artificialanalysis.ai/evaluations/scicode"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/scicode", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/scicode", "title": "SciCode"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-tau2-bench-telecom", "domain": "agentic", "name": "τ²-Bench Telecom", "released": "2025-06-12", "url": "https://artificialanalysis.ai/evaluations/tau2-bench"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/tau2-bench", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/tau2-bench", "title": "τ²-Bench Telecom"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-tau3-banking", "domain": "intelligence-index", "name": "τ³-Banking", "released": null, "url": "https://artificialanalysis.ai/evaluations/tau3-banking"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/tau3-banking", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/tau3-banking", "title": "τ³-Banking"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-terminal-bench-hard", "domain": "coding", "name": "Terminal-Bench Hard", "released": "2025-09-02", "url": "https://artificialanalysis.ai/evaluations/terminalbench-hard"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/terminalbench-hard", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-hard", "title": "Terminal-Bench Hard"}, {"benchmarks": [{"benchmark_id": "artificial-analysis-terminal-bench-v2-1", "domain": "intelligence-index", "name": "Terminal-Bench v2.1", "released": "2026-05-06", "url": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1"}], "document_type": "registry_page", "id": "artificial_analysis:https://artificialanalysis.ai/evaluations/terminalbench-v2-1", "source": "artificial_analysis", "source_url": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1", "title": "Terminal-Bench v2.1"}, {"benchmarks": [{"benchmark_id": "llm-stats-aa-briefcase", "domain": "productivity", "name": "AA-Briefcase", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500", "title": "AA-Briefcase"}, {"benchmarks": [{"benchmark_id": "llm-stats-aa-index", "domain": "general", "name": "AA-Index", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500", "title": "AA-Index"}, {"benchmarks": [{"benchmark_id": "llm-stats-aa-lcr", "domain": "long_context", "name": "AA-LCR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500", "title": "AA-LCR"}, {"benchmarks": [{"benchmark_id": "llm-stats-aa-omniscience-index", "domain": "reasoning", "name": "AA-Omniscience Index", "released": "2025-11-16", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500", "title": "AA-Omniscience Index"}, {"benchmarks": [{"benchmark_id": "llm-stats-acebench", "domain": "reasoning", "name": "ACEBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500", "title": "ACEBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-activitynet", "domain": "video", "name": "ActivityNet", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500", "title": "ActivityNet"}, {"benchmarks": [{"benchmark_id": "llm-stats-advancedif", "domain": "reasoning", "name": "AdvancedIF", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500", "title": "AdvancedIF"}, {"benchmarks": [{"benchmark_id": "llm-stats-aethercode", "domain": "reasoning", "name": "AetherCode", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500", "title": "AetherCode"}, {"benchmarks": [{"benchmark_id": "llm-stats-agent-startup-bench", "domain": "reasoning", "name": "Agent Startup Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500", "title": "Agent Startup Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-agents-last-exam", "domain": "reasoning", "name": "Agents' Last Exam", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500", "title": "Agents' Last Exam"}, {"benchmarks": [{"benchmark_id": "llm-stats-agieval", "domain": "legal", "name": "AGIEval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500", "title": "AGIEval"}, {"benchmarks": [{"benchmark_id": "llm-stats-ai2-reasoning-challenge-arc", "domain": "reasoning", "name": "AI2 Reasoning Challenge (ARC)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500", "title": "AI2 Reasoning Challenge (ARC)"}, {"benchmarks": [{"benchmark_id": "llm-stats-ai2d", "domain": "multimodal", "name": "AI2D", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500", "title": "AI2D"}, {"benchmarks": [{"benchmark_id": "llm-stats-aider-polyglot-edit", "domain": "general", "name": "Aider-Polyglot Edit", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500", "title": "Aider-Polyglot Edit"}, {"benchmarks": [{"benchmark_id": "llm-stats-aider-polyglot", "domain": "general", "name": "Aider-Polyglot", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500", "title": "Aider-Polyglot"}, {"benchmarks": [{"benchmark_id": "llm-stats-aider", "domain": "reasoning", "name": "Aider", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500", "title": "Aider"}, {"benchmarks": [{"benchmark_id": "llm-stats-aime-2024", "domain": "math", "name": "AIME 2024", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500", "title": "AIME 2024"}, {"benchmarks": [{"benchmark_id": "llm-stats-aime-2025", "domain": "math", "name": "AIME 2025", "released": "2025-02-12", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500", "title": "AIME 2025"}, {"benchmarks": [{"benchmark_id": "llm-stats-aime-2026", "domain": "math", "name": "AIME 2026", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500", "title": "AIME 2026"}, {"benchmarks": [{"benchmark_id": "llm-stats-aime", "domain": "math", "name": "AIME", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500", "title": "AIME"}, {"benchmarks": [{"benchmark_id": "llm-stats-air-bench", "domain": "safety", "name": "AIR-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500", "title": "AIR-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-aitz-em", "domain": "multimodal", "name": "AITZ_EM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500", "title": "AITZ_EM"}, {"benchmarks": [{"benchmark_id": "llm-stats-alignbench", "domain": "math", "name": "AlignBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500", "title": "AlignBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-alpacaeval-2-0", "domain": "reasoning", "name": "AlpacaEval 2.0", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500", "title": "AlpacaEval 2.0"}, {"benchmarks": [{"benchmark_id": "llm-stats-amc-2022-23", "domain": "math", "name": "AMC_2022_23", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500", "title": "AMC_2022_23"}, {"benchmarks": [{"benchmark_id": "llm-stats-amo-bench", "domain": "math", "name": "AMO Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500", "title": "AMO Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-android-control-high-em", "domain": "multimodal", "name": "Android Control High_EM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500", "title": "Android Control High_EM"}, {"benchmarks": [{"benchmark_id": "llm-stats-android-control-low-em", "domain": "multimodal", "name": "Android Control Low_EM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500", "title": "Android Control Low_EM"}, {"benchmarks": [{"benchmark_id": "llm-stats-androidbench", "domain": "agents", "name": "AndroidBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500", "title": "AndroidBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-androidworld-sr", "domain": "multimodal", "name": "AndroidWorld_SR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500", "title": "AndroidWorld_SR"}, {"benchmarks": [{"benchmark_id": "llm-stats-androidworld", "domain": "agents", "name": "AndroidWorld", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500", "title": "AndroidWorld"}, {"benchmarks": [{"benchmark_id": "llm-stats-apex-agents", "domain": "reasoning", "name": "APEX-Agents", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500", "title": "APEX-Agents"}, {"benchmarks": [{"benchmark_id": "llm-stats-apex-swe", "domain": "agents", "name": "APEX-SWE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500", "title": "APEX-SWE"}, {"benchmarks": [{"benchmark_id": "llm-stats-apex", "domain": "reasoning", "name": "Apex", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500", "title": "Apex"}, {"benchmarks": [{"benchmark_id": "llm-stats-api-bank", "domain": "reasoning", "name": "API-Bank", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500", "title": "API-Bank"}, {"benchmarks": [{"benchmark_id": "llm-stats-arc-agi-3", "domain": "reasoning", "name": "ARC-AGI-3", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500", "title": "ARC-AGI-3"}, {"benchmarks": [{"benchmark_id": "llm-stats-arc-agi-v2", "domain": "reasoning", "name": "ARC-AGI v2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500", "title": "ARC-AGI v2"}, {"benchmarks": [{"benchmark_id": "llm-stats-arc-agi", "domain": "reasoning", "name": "ARC-AGI", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500", "title": "ARC-AGI"}, {"benchmarks": [{"benchmark_id": "llm-stats-arc-c", "domain": "reasoning", "name": "ARC-C", "released": "2018-03-14", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500", "title": "ARC-C"}, {"benchmarks": [{"benchmark_id": "llm-stats-arc-e", "domain": "reasoning", "name": "ARC-E", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500", "title": "ARC-E"}, {"benchmarks": [{"benchmark_id": "llm-stats-arc", "domain": "reasoning", "name": "Arc", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500", "title": "Arc"}, {"benchmarks": [{"benchmark_id": "llm-stats-arcagi2", "domain": "reasoning", "name": "ArcAGI2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500", "title": "ArcAGI2"}, {"benchmarks": [{"benchmark_id": "llm-stats-arena-hard-v2", "domain": "reasoning", "name": "Arena-Hard v2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500", "title": "Arena-Hard v2"}, {"benchmarks": [{"benchmark_id": "llm-stats-arena-hard", "domain": "reasoning", "name": "Arena Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500", "title": "Arena Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-arkitscenes", "domain": "spatial_reasoning", "name": "ARKitScenes", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500", "title": "ARKitScenes"}, {"benchmarks": [{"benchmark_id": "llm-stats-artifacts-bench", "domain": "frontend_development", "name": "Artifacts Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500", "title": "Artifacts Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-artificial-analysis", "domain": "general", "name": "Artificial Analysis", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500", "title": "Artificial Analysis"}, {"benchmarks": [{"benchmark_id": "llm-stats-arxivmath", "domain": "math", "name": "ArXivMath", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500", "title": "ArXivMath"}, {"benchmarks": [{"benchmark_id": "llm-stats-attaq", "domain": "safety", "name": "AttaQ", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500", "title": "AttaQ"}, {"benchmarks": [{"benchmark_id": "llm-stats-autologi", "domain": "reasoning", "name": "AutoLogi", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500", "title": "AutoLogi"}, {"benchmarks": [{"benchmark_id": "llm-stats-automationbench-aa", "domain": "reasoning", "name": "AutomationBench-AA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500", "title": "AutomationBench-AA"}, {"benchmarks": [{"benchmark_id": "llm-stats-automationbench", "domain": "reasoning", "name": "AutomationBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500", "title": "AutomationBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-babyvision", "domain": "multimodal", "name": "BabyVision", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500", "title": "BabyVision"}, {"benchmarks": [{"benchmark_id": "llm-stats-bankertoolbench", "domain": "finance", "name": "BankerToolBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500", "title": "BankerToolBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-bbh", "domain": "math", "name": "BBH", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500", "title": "BBH"}, {"benchmarks": [{"benchmark_id": "llm-stats-bc-vl", "domain": "multimodal", "name": "BC-VL", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500", "title": "BC-VL"}, {"benchmarks": [{"benchmark_id": "llm-stats-beam-128k", "domain": "long_context", "name": "Beam 128K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500", "title": "Beam 128K"}, {"benchmarks": [{"benchmark_id": "llm-stats-benchcad-with-python-tool", "domain": "multimodal", "name": "BenchCAD (with Python tool)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500", "title": "BenchCAD (with Python tool)"}, {"benchmarks": [{"benchmark_id": "llm-stats-benchcad", "domain": "multimodal", "name": "BenchCAD", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500", "title": "BenchCAD"}, {"benchmarks": [{"benchmark_id": "llm-stats-beyond-aime", "domain": "math", "name": "Beyond AIME", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500", "title": "Beyond AIME"}, {"benchmarks": [{"benchmark_id": "llm-stats-bfcl-v2", "domain": "reasoning", "name": "BFCL v2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500", "title": "BFCL v2"}, {"benchmarks": [{"benchmark_id": "llm-stats-bfcl-v3-multiturn", "domain": "reasoning", "name": "BFCL_v3_MultiTurn", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500", "title": "BFCL_v3_MultiTurn"}, {"benchmarks": [{"benchmark_id": "llm-stats-bfcl-v3", "domain": "reasoning", "name": "BFCL-v3", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500", "title": "BFCL-v3"}, {"benchmarks": [{"benchmark_id": "llm-stats-bfcl-v4", "domain": "agents", "name": "BFCL-V4", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500", "title": "BFCL-V4"}, {"benchmarks": [{"benchmark_id": "llm-stats-bfcl", "domain": "reasoning", "name": "BFCL", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500", "title": "BFCL"}, {"benchmarks": [{"benchmark_id": "llm-stats-big-bench-audio", "domain": "reasoning", "name": "Big Bench Audio", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500", "title": "Big Bench Audio"}, {"benchmarks": [{"benchmark_id": "llm-stats-big-bench-extra-hard", "domain": "reasoning", "name": "BIG-Bench Extra Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500", "title": "BIG-Bench Extra Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-big-bench-hard", "domain": "math", "name": "BIG-Bench Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500", "title": "BIG-Bench Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-big-bench", "domain": "math", "name": "BIG-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500", "title": "BIG-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-big-finance-bench", "domain": "reasoning", "name": "Big Finance Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500", "title": "Big Finance Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-bigcodebench-full", "domain": "reasoning", "name": "BigCodeBench-Full", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500", "title": "BigCodeBench-Full"}, {"benchmarks": [{"benchmark_id": "llm-stats-bigcodebench-hard", "domain": "reasoning", "name": "BigCodeBench-Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500", "title": "BigCodeBench-Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-bigcodebench", "domain": "reasoning", "name": "BigCodeBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500", "title": "BigCodeBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-biolp-bench", "domain": "safety", "name": "BioLP-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500", "title": "BioLP-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-biomysterybench", "domain": "reasoning", "name": "BioMysteryBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500", "title": "BioMysteryBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-bird-sql-dev", "domain": "reasoning", "name": "Bird-SQL (dev)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500", "title": "Bird-SQL (dev)"}, {"benchmarks": [{"benchmark_id": "llm-stats-bixbench", "domain": "reasoning", "name": "BixBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500", "title": "BixBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-blink", "domain": "multimodal", "name": "BLINK", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500", "title": "BLINK"}, {"benchmarks": [{"benchmark_id": "llm-stats-blueprint-bench-2", "domain": "multimodal", "name": "Blueprint-Bench 2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500", "title": "Blueprint-Bench 2"}, {"benchmarks": [{"benchmark_id": "llm-stats-boolq", "domain": "reasoning", "name": "BoolQ", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500", "title": "BoolQ"}, {"benchmarks": [{"benchmark_id": "llm-stats-browsecomp-long-128k", "domain": "reasoning", "name": "BrowseComp Long Context 128k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500", "title": "BrowseComp Long Context 128k"}, {"benchmarks": [{"benchmark_id": "llm-stats-browsecomp-long-256k", "domain": "reasoning", "name": "BrowseComp Long Context 256k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500", "title": "BrowseComp Long Context 256k"}, {"benchmarks": [{"benchmark_id": "llm-stats-browsecomp-vl", "domain": "multimodal", "name": "BrowseComp-VL", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500", "title": "BrowseComp-VL"}, {"benchmarks": [{"benchmark_id": "llm-stats-browsecomp-zh", "domain": "reasoning", "name": "BrowseComp-zh", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500", "title": "BrowseComp-zh"}, {"benchmarks": [{"benchmark_id": "llm-stats-browsecomp", "domain": "reasoning", "name": "BrowseComp", "released": "2025-04-10", "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500", "title": "BrowseComp"}, {"benchmarks": [{"benchmark_id": "llm-stats-c-eval", "domain": "reasoning", "name": "C-Eval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500", "title": "C-Eval"}, {"benchmarks": [{"benchmark_id": "llm-stats-capture-the-flag-challenges", "domain": "safety", "name": "Capture-the-Flag Challenges (Internal)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500", "title": "Capture-the-Flag Challenges (Internal)"}, {"benchmarks": [{"benchmark_id": "llm-stats-cbnsl", "domain": "math", "name": "CBNSL", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500", "title": "CBNSL"}, {"benchmarks": [{"benchmark_id": "llm-stats-cc-bench-v2-backend", "domain": "code", "name": "CC-Bench-V2 Backend", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500", "title": "CC-Bench-V2 Backend"}, {"benchmarks": [{"benchmark_id": "llm-stats-cc-bench-v2-frontend", "domain": "code", "name": "CC-Bench-V2 Frontend", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500", "title": "CC-Bench-V2 Frontend"}, {"benchmarks": [{"benchmark_id": "llm-stats-cc-bench-v2-repo", "domain": "agents", "name": "CC-Bench-V2 Repo Exploration", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500", "title": "CC-Bench-V2 Repo Exploration"}, {"benchmarks": [{"benchmark_id": "llm-stats-cc-ocr", "domain": "multimodal", "name": "CC-OCR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500", "title": "CC-OCR"}, {"benchmarks": [{"benchmark_id": "llm-stats-cfeval", "domain": "code", "name": "CFEval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500", "title": "CFEval"}, {"benchmarks": [{"benchmark_id": "llm-stats-charadessta", "domain": "multimodal", "name": "CharadesSTA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500", "title": "CharadesSTA"}, {"benchmarks": [{"benchmark_id": "llm-stats-chartmuseum", "domain": "multimodal", "name": "ChartMuseum", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500", "title": "ChartMuseum"}, {"benchmarks": [{"benchmark_id": "llm-stats-chartqa", "domain": "multimodal", "name": "ChartQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500", "title": "ChartQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-chartqapro", "domain": "multimodal", "name": "ChartQAPro", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500", "title": "ChartQAPro"}, {"benchmarks": [{"benchmark_id": "llm-stats-charxiv-d", "domain": "multimodal", "name": "CharXiv-D", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500", "title": "CharXiv-D"}, {"benchmarks": [{"benchmark_id": "llm-stats-charxiv-r", "domain": "multimodal", "name": "CharXiv-R", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500", "title": "CharXiv-R"}, {"benchmarks": [{"benchmark_id": "llm-stats-chexpert-cxr", "domain": "healthcare", "name": "CheXpert CXR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500", "title": "CheXpert CXR"}, {"benchmarks": [{"benchmark_id": "llm-stats-ci-memories-coverage", "domain": "memory", "name": "CI Memories Coverage", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500", "title": "CI Memories Coverage"}, {"benchmarks": [{"benchmark_id": "llm-stats-ci-memories-violation", "domain": "memory", "name": "CI Memories Violation Rate", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500", "title": "CI Memories Violation Rate"}, {"benchmarks": [{"benchmark_id": "llm-stats-cl-bench-life", "domain": "agents", "name": "CL-bench (Life)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500", "title": "CL-bench (Life)"}, {"benchmarks": [{"benchmark_id": "llm-stats-cl-bench", "domain": "agents", "name": "CL-bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500", "title": "CL-bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-claw-eval-mm", "domain": "multimodal", "name": "ClawEval-MM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500", "title": "ClawEval-MM"}, {"benchmarks": [{"benchmark_id": "llm-stats-claw-eval", "domain": "agents", "name": "Claw-Eval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500", "title": "Claw-Eval"}, {"benchmarks": [{"benchmark_id": "llm-stats-cloningscenarios", "domain": "reasoning", "name": "CloningScenarios", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500", "title": "CloningScenarios"}, {"benchmarks": [{"benchmark_id": "llm-stats-cluewsc", "domain": "reasoning", "name": "CLUEWSC", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500", "title": "CLUEWSC"}, {"benchmarks": [{"benchmark_id": "llm-stats-cmmlu", "domain": "reasoning", "name": "CMMLU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500", "title": "CMMLU"}, {"benchmarks": [{"benchmark_id": "llm-stats-cmt-benchmark", "domain": "physics", "name": "CMT-Benchmark", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500", "title": "CMT-Benchmark"}, {"benchmarks": [{"benchmark_id": "llm-stats-cnmo-2024", "domain": "math", "name": "CNMO 2024", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500", "title": "CNMO 2024"}, {"benchmarks": [{"benchmark_id": "llm-stats-codeforces", "domain": "math", "name": "CodeForces", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500", "title": "CodeForces"}, {"benchmarks": [{"benchmark_id": "llm-stats-codegolf-v2-2", "domain": "code", "name": "Codegolf v2.2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500", "title": "Codegolf v2.2"}, {"benchmarks": [{"benchmark_id": "llm-stats-cohere-agentic-question-answering", "domain": "question_answering", "name": "Cohere Agentic Question Answering", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500", "title": "Cohere Agentic Question Answering"}, {"benchmarks": [{"benchmark_id": "llm-stats-cohere-data-analysis", "domain": "reasoning", "name": "Cohere Data Analysis", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500", "title": "Cohere Data Analysis"}, {"benchmarks": [{"benchmark_id": "llm-stats-cohere-memory-usage-quality", "domain": "memory", "name": "Cohere Memory Usage Quality", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500", "title": "Cohere Memory Usage Quality"}, {"benchmarks": [{"benchmark_id": "llm-stats-collie", "domain": "reasoning", "name": "COLLIE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500", "title": "COLLIE"}, {"benchmarks": [{"benchmark_id": "llm-stats-common-voice-15", "domain": "speech_to_text", "name": "Common Voice 15", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500", "title": "Common Voice 15"}, {"benchmarks": [{"benchmark_id": "llm-stats-commonsenseqa", "domain": "reasoning", "name": "CommonSenseQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500", "title": "CommonSenseQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-community-07c9946d-dcf0-4977-a640-a6b1356b4f0b", "domain": "general", "name": "independence-bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A07c9946d-dcf0-4977-a640-a6b1356b4f0b?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A07c9946d-dcf0-4977-a640-a6b1356b4f0b?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A07c9946d-dcf0-4977-a640-a6b1356b4f0b?top_n=500", "title": "independence-bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-community-2256e9c9-b256-4444-b639-7cc3b1855d96", "domain": "long_context", "name": "nolima", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A2256e9c9-b256-4444-b639-7cc3b1855d96?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A2256e9c9-b256-4444-b639-7cc3b1855d96?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A2256e9c9-b256-4444-b639-7cc3b1855d96?top_n=500", "title": "nolima"}, {"benchmarks": [{"benchmark_id": "llm-stats-community-5f95f778-c521-43fa-b80e-6a55465601e3", "domain": "other", "name": "ael_gate_benchmark_cases_template", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A5f95f778-c521-43fa-b80e-6a55465601e3?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A5f95f778-c521-43fa-b80e-6a55465601e3?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A5f95f778-c521-43fa-b80e-6a55465601e3?top_n=500", "title": "ael_gate_benchmark_cases_template"}, {"benchmarks": [{"benchmark_id": "llm-stats-community-64d67847-06bd-423a-923c-c2acfab82281", "domain": "other", "name": "OptimBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A64d67847-06bd-423a-923c-c2acfab82281?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A64d67847-06bd-423a-923c-c2acfab82281?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A64d67847-06bd-423a-923c-c2acfab82281?top_n=500", "title": "OptimBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-community-ed90e889-4678-4fbd-98ab-0e654f4bf35e", "domain": "other", "name": "atlas", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Aed90e889-4678-4fbd-98ab-0e654f4bf35e?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3Aed90e889-4678-4fbd-98ab-0e654f4bf35e?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Aed90e889-4678-4fbd-98ab-0e654f4bf35e?top_n=500", "title": "atlas"}, {"benchmarks": [{"benchmark_id": "llm-stats-community-fd462fc2-283c-4967-bd7d-b39d7c661807", "domain": "other", "name": "testing", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Afd462fc2-283c-4967-bd7d-b39d7c661807?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3Afd462fc2-283c-4967-bd7d-b39d7c661807?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Afd462fc2-283c-4967-bd7d-b39d7c661807?top_n=500", "title": "testing"}, {"benchmarks": [{"benchmark_id": "llm-stats-complexfuncbench", "domain": "long_context", "name": "ComplexFuncBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500", "title": "ComplexFuncBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-contphy", "domain": "multimodal", "name": "ContPhy", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500", "title": "ContPhy"}, {"benchmarks": [{"benchmark_id": "llm-stats-corpusqa-1m", "domain": "long_context", "name": "CorpusQA 1M", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500", "title": "CorpusQA 1M"}, {"benchmarks": [{"benchmark_id": "llm-stats-corpusqa", "domain": "long_context", "name": "CorpusQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500", "title": "CorpusQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-countbench", "domain": "reasoning", "name": "CountBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500", "title": "CountBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-countqa", "domain": "multimodal", "name": "CountQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500", "title": "CountQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-covost2-en-zh", "domain": "speech_to_text", "name": "CoVoST2 en-zh", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500", "title": "CoVoST2 en-zh"}, {"benchmarks": [{"benchmark_id": "llm-stats-covost2", "domain": "speech_to_text", "name": "CoVoST2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500", "title": "CoVoST2"}, {"benchmarks": [{"benchmark_id": "llm-stats-coworkbench", "domain": "productivity", "name": "CoWorkBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500", "title": "CoWorkBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-crag", "domain": "reasoning", "name": "CRAG", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500", "title": "CRAG"}, {"benchmarks": [{"benchmark_id": "llm-stats-creative-writing-v3", "domain": "creativity", "name": "Creative Writing v3", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500", "title": "Creative Writing v3"}, {"benchmarks": [{"benchmark_id": "llm-stats-creativework", "domain": "reasoning", "name": "CreativeWork", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500", "title": "CreativeWork"}, {"benchmarks": [{"benchmark_id": "llm-stats-critpt", "domain": "math", "name": "CritPT", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500", "title": "CritPT"}, {"benchmarks": [{"benchmark_id": "llm-stats-crossvid", "domain": "multimodal", "name": "CrossVid", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500", "title": "CrossVid"}, {"benchmarks": [{"benchmark_id": "llm-stats-crperelation", "domain": "reasoning", "name": "CRPErelation", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500", "title": "CRPErelation"}, {"benchmarks": [{"benchmark_id": "llm-stats-crux-o", "domain": "reasoning", "name": "CRUX-O", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500", "title": "CRUX-O"}, {"benchmarks": [{"benchmark_id": "llm-stats-cruxeval-input-cot", "domain": "reasoning", "name": "CRUXEval-Input-CoT", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500", "title": "CRUXEval-Input-CoT"}, {"benchmarks": [{"benchmark_id": "llm-stats-cruxeval-o", "domain": "reasoning", "name": "CruxEval-O", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500", "title": "CruxEval-O"}, {"benchmarks": [{"benchmark_id": "llm-stats-cruxeval-output-cot", "domain": "reasoning", "name": "CRUXEval-Output-CoT", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500", "title": "CRUXEval-Output-CoT"}, {"benchmarks": [{"benchmark_id": "llm-stats-csimpleqa", "domain": "language", "name": "CSimpleQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500", "title": "CSimpleQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-cursorbench-3-2", "domain": "agents", "name": "CursorBench v3.2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500", "title": "CursorBench v3.2"}, {"benchmarks": [{"benchmark_id": "llm-stats-cvtg-2k", "domain": "image-generation", "name": "CVTG-2K", "released": "2025-03-30", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cvtg-2k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cvtg-2k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cvtg-2k?top_n=500", "title": "CVTG-2K"}, {"benchmarks": [{"benchmark_id": "llm-stats-cybench", "domain": "safety", "name": "CyBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500", "title": "CyBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-cybergym", "domain": "safety", "name": "CyberGym", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500", "title": "CyberGym"}, {"benchmarks": [{"benchmark_id": "llm-stats-cyberseceval-4", "domain": "safety", "name": "CyberSecEval 4", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500", "title": "CyberSecEval 4"}, {"benchmarks": [{"benchmark_id": "llm-stats-cybersecurity-ctfs", "domain": "safety", "name": "Cybersecurity CTFs", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500", "title": "Cybersecurity CTFs"}, {"benchmarks": [{"benchmark_id": "llm-stats-dailyomni", "domain": "multimodal", "name": "DailyOmni", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500", "title": "DailyOmni"}, {"benchmarks": [{"benchmark_id": "llm-stats-deck-bench", "domain": "productivity", "name": "DECK-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500", "title": "DECK-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-deep-planning", "domain": "reasoning", "name": "DeepPlanning", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500", "title": "DeepPlanning"}, {"benchmarks": [{"benchmark_id": "llm-stats-deepsearchqa", "domain": "reasoning", "name": "DeepSearchQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500", "title": "DeepSearchQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-deepswe-1-0", "domain": "agents", "name": "DeepSWE 1.0", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500", "title": "DeepSWE 1.0"}, {"benchmarks": [{"benchmark_id": "llm-stats-deepswe-1-1", "domain": "agents", "name": "DeepSWE 1.1", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500", "title": "DeepSWE 1.1"}, {"benchmarks": [{"benchmark_id": "llm-stats-deepswe", "domain": "agents", "name": "DeepSWE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500", "title": "DeepSWE"}, {"benchmarks": [{"benchmark_id": "llm-stats-dermmcqa", "domain": "healthcare", "name": "DermMCQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500", "title": "DermMCQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-design2code", "domain": "multimodal", "name": "Design2Code", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500", "title": "Design2Code"}, {"benchmarks": [{"benchmark_id": "llm-stats-docvqa", "domain": "multimodal", "name": "DocVQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500", "title": "DocVQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-docvqatest", "domain": "multimodal", "name": "DocVQAtest", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500", "title": "DocVQAtest"}, {"benchmarks": [{"benchmark_id": "llm-stats-doubao-multi-turn-bench", "domain": "instruction_following", "name": "Doubao Multi-Turn Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500", "title": "Doubao Multi-Turn Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-draco", "domain": "reasoning", "name": "DRACO", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500", "title": "DRACO"}, {"benchmarks": [{"benchmark_id": "llm-stats-drop", "domain": "math", "name": "DROP", "released": "2019-03-01", "url": "https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500", "title": "DROP"}, {"benchmarks": [{"benchmark_id": "llm-stats-ds-arena-code", "domain": "reasoning", "name": "DS-Arena-Code", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500", "title": "DS-Arena-Code"}, {"benchmarks": [{"benchmark_id": "llm-stats-ds-fim-eval", "domain": "general", "name": "DS-FIM-Eval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500", "title": "DS-FIM-Eval"}, {"benchmarks": [{"benchmark_id": "llm-stats-dsbench-fullstack", "domain": "agents", "name": "DSBench-FullStack", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500", "title": "DSBench-FullStack"}, {"benchmarks": [{"benchmark_id": "llm-stats-dsbench-hard", "domain": "agents", "name": "DSBench-Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500", "title": "DSBench-Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-dude", "domain": "long_context", "name": "DUDE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500", "title": "DUDE"}, {"benchmarks": [{"benchmark_id": "llm-stats-dynamath", "domain": "math", "name": "DynaMath", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500", "title": "DynaMath"}, {"benchmarks": [{"benchmark_id": "llm-stats-eclektic", "domain": "reasoning", "name": "ECLeKTic", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500", "title": "ECLeKTic"}, {"benchmarks": [{"benchmark_id": "llm-stats-egoschema", "domain": "long_context", "name": "EgoSchema", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500", "title": "EgoSchema"}, {"benchmarks": [{"benchmark_id": "llm-stats-embspatialbench", "domain": "spatial_reasoning", "name": "EmbSpatialBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500", "title": "EmbSpatialBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-emma", "domain": "math", "name": "EMMA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500", "title": "EMMA"}, {"benchmarks": [{"benchmark_id": "llm-stats-eq-bench", "domain": "reasoning", "name": "EQ-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500", "title": "EQ-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-erqa", "domain": "reasoning", "name": "ERQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500", "title": "ERQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-evalplus", "domain": "reasoning", "name": "EvalPlus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500", "title": "EvalPlus"}, {"benchmarks": [{"benchmark_id": "llm-stats-exploitbench", "domain": "safety", "name": "ExploitBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500", "title": "ExploitBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-exploitgym", "domain": "safety", "name": "ExploitGym", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500", "title": "ExploitGym"}, {"benchmarks": [{"benchmark_id": "llm-stats-facts-grounding", "domain": "reasoning", "name": "FACTS Grounding", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500", "title": "FACTS Grounding"}, {"benchmarks": [{"benchmark_id": "llm-stats-factscore", "domain": "reasoning", "name": "FActScore", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500", "title": "FActScore"}, {"benchmarks": [{"benchmark_id": "llm-stats-figqa", "domain": "safety", "name": "FigQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500", "title": "FigQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-finance-agent-v1-1", "domain": "reasoning", "name": "Finance Agent v1.1", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500", "title": "Finance Agent v1.1"}, {"benchmarks": [{"benchmark_id": "llm-stats-finance-agent-v2", "domain": "reasoning", "name": "Finance Agent v2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500", "title": "Finance Agent v2"}, {"benchmarks": [{"benchmark_id": "llm-stats-finance-agent", "domain": "reasoning", "name": "Finance Agent", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500", "title": "Finance Agent"}, {"benchmarks": [{"benchmark_id": "llm-stats-finqa", "domain": "math", "name": "FinQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500", "title": "FinQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-finsearchcomp-t2-t3", "domain": "reasoning", "name": "FinSearchComp T2&T3", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500", "title": "FinSearchComp T2&T3"}, {"benchmarks": [{"benchmark_id": "llm-stats-finsearchcomp-t3", "domain": "reasoning", "name": "FinSearchComp-T3", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500", "title": "FinSearchComp-T3"}, {"benchmarks": [{"benchmark_id": "llm-stats-flame-vlm-code", "domain": "multimodal", "name": "Flame-VLM-Code", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500", "title": "Flame-VLM-Code"}, {"benchmarks": [{"benchmark_id": "llm-stats-flenqa", "domain": "long_context", "name": "FlenQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500", "title": "FlenQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-fleurs", "domain": "speech_to_text", "name": "FLEURS", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500", "title": "FLEURS"}, {"benchmarks": [{"benchmark_id": "llm-stats-frames", "domain": "reasoning", "name": "FRAMES", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500", "title": "FRAMES"}, {"benchmarks": [{"benchmark_id": "llm-stats-french-mmlu", "domain": "legal", "name": "French MMLU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500", "title": "French MMLU"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontier-bench-v0-1", "domain": "agents", "name": "Frontier-Bench v0.1", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500", "title": "Frontier-Bench v0.1"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontier-science", "domain": "reasoning", "name": "Frontier Science", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500", "title": "Frontier Science"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontier-swe-impl", "domain": "agents", "name": "FrontierSWE (Impl.)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500", "title": "FrontierSWE (Impl.)"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontiercode-1-1", "domain": "reasoning", "name": "FrontierCode 1.1", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500", "title": "FrontierCode 1.1"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontiercode", "domain": "reasoning", "name": "FrontierCode", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500", "title": "FrontierCode"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontiercs", "domain": "reasoning", "name": "FrontierCS", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500", "title": "FrontierCS"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontiermath-tier-4-v2", "domain": "math", "name": "FrontierMath Tier 4 (v2)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500", "title": "FrontierMath Tier 4 (v2)"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontiermath", "domain": "math", "name": "FrontierMath", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500", "title": "FrontierMath"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontierscience-olympiad", "domain": "math", "name": "FrontierScience Olympiad", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500", "title": "FrontierScience Olympiad"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontierscience-research", "domain": "reasoning", "name": "FrontierScience Research", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500", "title": "FrontierScience Research"}, {"benchmarks": [{"benchmark_id": "llm-stats-frontierswe", "domain": "agents", "name": "FrontierSWE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500", "title": "FrontierSWE"}, {"benchmarks": [{"benchmark_id": "llm-stats-fullstackbench-en", "domain": "reasoning", "name": "FullStackBench en", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500", "title": "FullStackBench en"}, {"benchmarks": [{"benchmark_id": "llm-stats-fullstackbench-zh", "domain": "reasoning", "name": "FullStackBench zh", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500", "title": "FullStackBench zh"}, {"benchmarks": [{"benchmark_id": "llm-stats-functionalmath", "domain": "math", "name": "FunctionalMATH", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500", "title": "FunctionalMATH"}, {"benchmarks": [{"benchmark_id": "llm-stats-gaia2", "domain": "reasoning", "name": "GAIA2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500", "title": "GAIA2"}, {"benchmarks": [{"benchmark_id": "llm-stats-gameworld", "domain": "multimodal", "name": "GameWorld", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500", "title": "GameWorld"}, {"benchmarks": [{"benchmark_id": "llm-stats-gdp-pdf", "domain": "multimodal", "name": "GDP.pdf", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500", "title": "GDP.pdf"}, {"benchmarks": [{"benchmark_id": "llm-stats-gdpval-aa", "domain": "legal", "name": "GDPval-AA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500", "title": "GDPval-AA"}, {"benchmarks": [{"benchmark_id": "llm-stats-gdpval-mm", "domain": "multimodal", "name": "GDPval-MM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500", "title": "GDPval-MM"}, {"benchmarks": [{"benchmark_id": "llm-stats-gdpval-rubrics", "domain": "legal", "name": "GDPval-Rubrics", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500", "title": "GDPval-Rubrics"}, {"benchmarks": [{"benchmark_id": "llm-stats-gdpval", "domain": "legal", "name": "GDPval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500", "title": "GDPval"}, {"benchmarks": [{"benchmark_id": "llm-stats-genebench-pro", "domain": "reasoning", "name": "GeneBench-Pro", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500", "title": "GeneBench-Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-genebench", "domain": "reasoning", "name": "GeneBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500", "title": "GeneBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-giantsteps-tempo", "domain": "audio", "name": "GiantSteps Tempo", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500", "title": "GiantSteps Tempo"}, {"benchmarks": [{"benchmark_id": "llm-stats-global-mmlu-lite", "domain": "reasoning", "name": "Global-MMLU-Lite", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500", "title": "Global-MMLU-Lite"}, {"benchmarks": [{"benchmark_id": "llm-stats-global-mmlu", "domain": "reasoning", "name": "Global-MMLU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500", "title": "Global-MMLU"}, {"benchmarks": [{"benchmark_id": "llm-stats-global-piqa", "domain": "physics", "name": "Global PIQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500", "title": "Global PIQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-gorilla-benchmark-api-bench", "domain": "reasoning", "name": "Gorilla Benchmark API Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500", "title": "Gorilla Benchmark API Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-govreport", "domain": "long_context", "name": "GovReport", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500", "title": "GovReport"}, {"benchmarks": [{"benchmark_id": "llm-stats-gpqa-biology", "domain": "reasoning", "name": "GPQA Biology", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500", "title": "GPQA Biology"}, {"benchmarks": [{"benchmark_id": "llm-stats-gpqa-chemistry", "domain": "reasoning", "name": "GPQA Chemistry", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500", "title": "GPQA Chemistry"}, {"benchmarks": [{"benchmark_id": "llm-stats-gpqa-physics", "domain": "physics", "name": "GPQA Physics", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500", "title": "GPQA Physics"}, {"benchmarks": [{"benchmark_id": "llm-stats-gpqa", "domain": "physics", "name": "GPQA", "released": "2023-11-20", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500", "title": "GPQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-graphwalks-bfs-128k", "domain": "reasoning", "name": "Graphwalks BFS <128k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500", "title": "Graphwalks BFS <128k"}, {"benchmarks": [{"benchmark_id": "llm-stats-graphwalks-bfs-128k-2", "domain": "long_context", "name": "Graphwalks BFS >128k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500", "title": "Graphwalks BFS >128k"}, {"benchmarks": [{"benchmark_id": "llm-stats-graphwalks-bfs-1m", "domain": "long_context", "name": "Graphwalks BFS 1M", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500", "title": "Graphwalks BFS 1M"}, {"benchmarks": [{"benchmark_id": "llm-stats-graphwalks-parents-128k", "domain": "reasoning", "name": "Graphwalks parents <128k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500", "title": "Graphwalks parents <128k"}, {"benchmarks": [{"benchmark_id": "llm-stats-graphwalks-parents-128k-2", "domain": "long_context", "name": "Graphwalks parents >128k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500", "title": "Graphwalks parents >128k"}, {"benchmarks": [{"benchmark_id": "llm-stats-graphwalks", "domain": "long_context", "name": "GraphWalks", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500", "title": "GraphWalks"}, {"benchmarks": [{"benchmark_id": "llm-stats-groundui-1k", "domain": "multimodal", "name": "GroundUI-1K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500", "title": "GroundUI-1K"}, {"benchmarks": [{"benchmark_id": "llm-stats-gsm-8k-cot", "domain": "math", "name": "GSM-8K (CoT)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500", "title": "GSM-8K (CoT)"}, {"benchmarks": [{"benchmark_id": "llm-stats-gsm8k-chat", "domain": "math", "name": "GSM8K Chat", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500", "title": "GSM8K Chat"}, {"benchmarks": [{"benchmark_id": "llm-stats-gsm8k", "domain": "math", "name": "GSM8k", "released": "2021-04-01", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500", "title": "GSM8k"}, {"benchmarks": [{"benchmark_id": "llm-stats-hallusion-bench", "domain": "reasoning", "name": "Hallusion Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500", "title": "Hallusion Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-harvey-lab-aa", "domain": "knowledge", "name": "Harvey LAB-AA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500", "title": "Harvey LAB-AA"}, {"benchmarks": [{"benchmark_id": "llm-stats-harvey-lab", "domain": "knowledge", "name": "Harvey LAB (Vals)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500", "title": "Harvey LAB (Vals)"}, {"benchmarks": [{"benchmark_id": "llm-stats-healthbench-consensus", "domain": "healthcare", "name": "HealthBench Consensus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500", "title": "HealthBench Consensus"}, {"benchmarks": [{"benchmark_id": "llm-stats-healthbench-hard", "domain": "healthcare", "name": "HealthBench Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500", "title": "HealthBench Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-healthbench-professional", "domain": "healthcare", "name": "HealthBench Professional", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500", "title": "HealthBench Professional"}, {"benchmarks": [{"benchmark_id": "llm-stats-healthbench", "domain": "healthcare", "name": "HealthBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500", "title": "HealthBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-hellaswag", "domain": "reasoning", "name": "HellaSwag", "released": "2019-05-19", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500", "title": "HellaSwag"}, {"benchmarks": [{"benchmark_id": "llm-stats-hiddenmath", "domain": "math", "name": "HiddenMath", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500", "title": "HiddenMath"}, {"benchmarks": [{"benchmark_id": "llm-stats-hipho", "domain": "multimodal", "name": "HiPhO", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500", "title": "HiPhO"}, {"benchmarks": [{"benchmark_id": "llm-stats-hle-verified", "domain": "reasoning", "name": "HLE-Verified", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500", "title": "HLE-Verified"}, {"benchmarks": [{"benchmark_id": "llm-stats-hmmt-2025", "domain": "math", "name": "HMMT 2025", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500", "title": "HMMT 2025"}, {"benchmarks": [{"benchmark_id": "llm-stats-hmmt-feb-26", "domain": "math", "name": "HMMT Feb 26", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500", "title": "HMMT Feb 26"}, {"benchmarks": [{"benchmark_id": "llm-stats-hmmt25", "domain": "math", "name": "HMMT25", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500", "title": "HMMT25"}, {"benchmarks": [{"benchmark_id": "llm-stats-horizonmath", "domain": "math", "name": "HorizonMath", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500", "title": "HorizonMath"}, {"benchmarks": [{"benchmark_id": "llm-stats-hr-bench-4k", "domain": "multimodal", "name": "HR-Bench (4k)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500", "title": "HR-Bench (4k)"}, {"benchmarks": [{"benchmark_id": "llm-stats-humaneval-2", "domain": "reasoning", "name": "HumanEval+", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500", "title": "HumanEval+"}, {"benchmarks": [{"benchmark_id": "llm-stats-humaneval-average", "domain": "reasoning", "name": "HumanEval-Average", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500", "title": "HumanEval-Average"}, {"benchmarks": [{"benchmark_id": "llm-stats-humaneval-er", "domain": "reasoning", "name": "HumanEval-ER", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500", "title": "HumanEval-ER"}, {"benchmarks": [{"benchmark_id": "llm-stats-humaneval-mul", "domain": "reasoning", "name": "HumanEval-Mul", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500", "title": "HumanEval-Mul"}, {"benchmarks": [{"benchmark_id": "llm-stats-humaneval-plus", "domain": "reasoning", "name": "HumanEval Plus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500", "title": "HumanEval Plus"}, {"benchmarks": [{"benchmark_id": "llm-stats-humaneval", "domain": "reasoning", "name": "HumanEval", "released": "2021-07-08", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500", "title": "HumanEval"}, {"benchmarks": [{"benchmark_id": "llm-stats-humanevalfim-average", "domain": "general", "name": "HumanEvalFIM-Average", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500", "title": "HumanEvalFIM-Average"}, {"benchmarks": [{"benchmark_id": "llm-stats-humanity-s-last-exam-no-tools-text-only", "domain": "math", "name": "Humanity's Last Exam (no tools, text-only)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500", "title": "Humanity's Last Exam (no tools, text-only)"}, {"benchmarks": [{"benchmark_id": "llm-stats-humanity-s-last-exam-with-tools-text-only", "domain": "math", "name": "Humanity's Last Exam (with tools, text-only)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500", "title": "Humanity's Last Exam (with tools, text-only)"}, {"benchmarks": [{"benchmark_id": "llm-stats-humanity-s-last-exam", "domain": "math", "name": "Humanity's Last Exam", "released": "2025-01-24", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500", "title": "Humanity's Last Exam"}, {"benchmarks": [{"benchmark_id": "llm-stats-hypersim", "domain": "spatial_reasoning", "name": "Hypersim", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500", "title": "Hypersim"}, {"benchmarks": [{"benchmark_id": "llm-stats-if", "domain": "structured_output", "name": "IF", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500", "title": "IF"}, {"benchmarks": [{"benchmark_id": "llm-stats-ifbench", "domain": "instruction_following", "name": "IFBench", "released": "2025-07-03", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500", "title": "IFBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-ifeval", "domain": "structured_output", "name": "IFEval", "released": "2023-11-14", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500", "title": "IFEval"}, {"benchmarks": [{"benchmark_id": "llm-stats-image2floorplan", "domain": "multimodal", "name": "Image2FloorPlan", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500", "title": "Image2FloorPlan"}, {"benchmarks": [{"benchmark_id": "llm-stats-imagemining", "domain": "multimodal", "name": "ImageMining", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500", "title": "ImageMining"}, {"benchmarks": [{"benchmark_id": "llm-stats-imo-2025", "domain": "math", "name": "IMO 2025", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500", "title": "IMO 2025"}, {"benchmarks": [{"benchmark_id": "llm-stats-imo-answerbench", "domain": "math", "name": "IMO-AnswerBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500", "title": "IMO-AnswerBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-imoproof-adv", "domain": "math", "name": "IMOProof-Adv", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500", "title": "IMOProof-Adv"}, {"benchmarks": [{"benchmark_id": "llm-stats-include", "domain": "general", "name": "Include", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500", "title": "Include"}, {"benchmarks": [{"benchmark_id": "llm-stats-infinitebench-en-mc", "domain": "long_context", "name": "InfiniteBench/En.MC", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500", "title": "InfiniteBench/En.MC"}, {"benchmarks": [{"benchmark_id": "llm-stats-infinitebench-en-qa", "domain": "long_context", "name": "InfiniteBench/En.QA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500", "title": "InfiniteBench/En.QA"}, {"benchmarks": [{"benchmark_id": "llm-stats-infographicsqa", "domain": "multimodal", "name": "InfographicsQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500", "title": "InfographicsQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-infovqa", "domain": "multimodal", "name": "InfoVQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500", "title": "InfoVQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-infovqatest", "domain": "multimodal", "name": "InfoVQAtest", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500", "title": "InfoVQAtest"}, {"benchmarks": [{"benchmark_id": "llm-stats-instruct-humaneval", "domain": "general", "name": "Instruct HumanEval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500", "title": "Instruct HumanEval"}, {"benchmarks": [{"benchmark_id": "llm-stats-intergps", "domain": "math", "name": "InterGPS", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500", "title": "InterGPS"}, {"benchmarks": [{"benchmark_id": "llm-stats-internal-api-instruction-following-hard", "domain": "structured_output", "name": "Internal API instruction following (hard)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500", "title": "Internal API instruction following (hard)"}, {"benchmarks": [{"benchmark_id": "llm-stats-internal-research-debugging-evaluation", "domain": "reasoning", "name": "Internal Research Debugging Evaluation", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500", "title": "Internal Research Debugging Evaluation"}, {"benchmarks": [{"benchmark_id": "llm-stats-ipho-2025", "domain": "physics", "name": "IPhO 2025", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500", "title": "IPhO 2025"}, {"benchmarks": [{"benchmark_id": "llm-stats-job-bench", "domain": "productivity", "name": "Job Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500", "title": "Job Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-kernel-bench-l3", "domain": "agents", "name": "Kernel Bench L3", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500", "title": "Kernel Bench L3"}, {"benchmarks": [{"benchmark_id": "llm-stats-kernelbench-hard", "domain": "agents", "name": "KernelBench Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500", "title": "KernelBench Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-kernelgen-1p", "domain": "agents", "name": "KernelGen 1P", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500", "title": "KernelGen 1P"}, {"benchmarks": [{"benchmark_id": "llm-stats-kimi-claw-24-7-bench", "domain": "agents", "name": "Kimi Claw 24/7 Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500", "title": "Kimi Claw 24/7 Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-kimi-code-bench-v2", "domain": "agents", "name": "Kimi Code Bench v2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500", "title": "Kimi Code Bench v2"}, {"benchmarks": [{"benchmark_id": "llm-stats-kina", "domain": "reasoning", "name": "KINA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500", "title": "KINA"}, {"benchmarks": [{"benchmark_id": "llm-stats-labbench2", "domain": "reasoning", "name": "LABBench2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500", "title": "LABBench2"}, {"benchmarks": [{"benchmark_id": "llm-stats-lbpp-v2", "domain": "reasoning", "name": "LBPP (v2)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500", "title": "LBPP (v2)"}, {"benchmarks": [{"benchmark_id": "llm-stats-legal-agent-benchmark", "domain": "legal", "name": "Legal Agent Benchmark", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500", "title": "Legal Agent Benchmark"}, {"benchmarks": [{"benchmark_id": "llm-stats-lifescibench", "domain": "multimodal", "name": "LifeSciBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500", "title": "LifeSciBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-lingoqa", "domain": "multimodal", "name": "LingoQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500", "title": "LingoQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-livebench-20241125", "domain": "math", "name": "LiveBench 20241125", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500", "title": "LiveBench 20241125"}, {"benchmarks": [{"benchmark_id": "llm-stats-livebench", "domain": "math", "name": "LiveBench", "released": "2024-06-12", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500", "title": "LiveBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-livecodebench-01-09", "domain": "reasoning", "name": "LiveCodeBench(01-09)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500", "title": "LiveCodeBench(01-09)"}, {"benchmarks": [{"benchmark_id": "llm-stats-livecodebench-pro", "domain": "reasoning", "name": "LiveCodeBench Pro", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500", "title": "LiveCodeBench Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-livecodebench-v5-24-12-25-2", "domain": "reasoning", "name": "LiveCodeBench v5 24.12-25.2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500", "title": "LiveCodeBench v5 24.12-25.2"}, {"benchmarks": [{"benchmark_id": "llm-stats-livecodebench-v5", "domain": "reasoning", "name": "LiveCodeBench v5", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500", "title": "LiveCodeBench v5"}, {"benchmarks": [{"benchmark_id": "llm-stats-livecodebench-v6", "domain": "reasoning", "name": "LiveCodeBench v6", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500", "title": "LiveCodeBench v6"}, {"benchmarks": [{"benchmark_id": "llm-stats-livecodebench", "domain": "reasoning", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500", "title": "LiveCodeBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-livemathematicianbench", "domain": "math", "name": "LiveMathematicianBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500", "title": "LiveMathematicianBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-livesports-3k", "domain": "multimodal", "name": "LiveSports-3K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500", "title": "LiveSports-3K"}, {"benchmarks": [{"benchmark_id": "llm-stats-livesqlbench", "domain": "agents", "name": "LiveSQLBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500", "title": "LiveSQLBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-lmarena-text", "domain": "reasoning", "name": "LMArena Text Leaderboard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500", "title": "LMArena Text Leaderboard"}, {"benchmarks": [{"benchmark_id": "llm-stats-loca-bench-256k", "domain": "reasoning", "name": "LOCA-Bench (256k)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500", "title": "LOCA-Bench (256k)"}, {"benchmarks": [{"benchmark_id": "llm-stats-longbench-v2", "domain": "long_context", "name": "LongBench v2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500", "title": "LongBench v2"}, {"benchmarks": [{"benchmark_id": "llm-stats-longcodebench", "domain": "long_context", "name": "LongCodeBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500", "title": "LongCodeBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-longfact-concepts", "domain": "reasoning", "name": "LongFact Concepts", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500", "title": "LongFact Concepts"}, {"benchmarks": [{"benchmark_id": "llm-stats-longfact-objects", "domain": "reasoning", "name": "LongFact Objects", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500", "title": "LongFact Objects"}, {"benchmarks": [{"benchmark_id": "llm-stats-longfact", "domain": "factuality", "name": "LongFact", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500", "title": "LongFact"}, {"benchmarks": [{"benchmark_id": "llm-stats-longtext-bench", "domain": "image-generation", "name": "LongText-Bench", "released": "2025-07-29", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longtext-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longtext-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longtext-bench?top_n=500", "title": "LongText-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-longvideobench", "domain": "long_context", "name": "LongVideoBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500", "title": "LongVideoBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-lsat", "domain": "legal", "name": "LSAT", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500", "title": "LSAT"}, {"benchmarks": [{"benchmark_id": "llm-stats-lvbench", "domain": "long_context", "name": "LVBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500", "title": "LVBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-management-consulting-tasks", "domain": "reasoning", "name": "Management Consulting Tasks (Internal)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500", "title": "Management Consulting Tasks (Internal)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mask", "domain": "reasoning", "name": "MASK", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500", "title": "MASK"}, {"benchmarks": [{"benchmark_id": "llm-stats-math-cot", "domain": "math", "name": "MATH (CoT)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500", "title": "MATH (CoT)"}, {"benchmarks": [{"benchmark_id": "llm-stats-math-500", "domain": "math", "name": "MATH-500", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500", "title": "MATH-500"}, {"benchmarks": [{"benchmark_id": "llm-stats-math", "domain": "math", "name": "MATH", "released": "2021-11-08", "url": "https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500", "title": "MATH"}, {"benchmarks": [{"benchmark_id": "llm-stats-matharena-apex", "domain": "math", "name": "MathArena Apex", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500", "title": "MathArena Apex"}, {"benchmarks": [{"benchmark_id": "llm-stats-mathverse-mini", "domain": "math", "name": "MathVerse-Mini", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500", "title": "MathVerse-Mini"}, {"benchmarks": [{"benchmark_id": "llm-stats-mathverse", "domain": "math", "name": "MathVerse", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500", "title": "MathVerse"}, {"benchmarks": [{"benchmark_id": "llm-stats-mathvision", "domain": "math", "name": "MathVision", "released": "2024-02-22", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500", "title": "MathVision"}, {"benchmarks": [{"benchmark_id": "llm-stats-mathvista-mini", "domain": "math", "name": "MathVista-Mini", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500", "title": "MathVista-Mini"}, {"benchmarks": [{"benchmark_id": "llm-stats-mathvista", "domain": "math", "name": "MathVista", "released": "2023-10-03", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500", "title": "MathVista"}, {"benchmarks": [{"benchmark_id": "llm-stats-maverix", "domain": "multimodal", "name": "MAVERIX", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500", "title": "MAVERIX"}, {"benchmarks": [{"benchmark_id": "llm-stats-maxife", "domain": "general", "name": "MAXIFE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500", "title": "MAXIFE"}, {"benchmarks": [{"benchmark_id": "llm-stats-mbpp-2", "domain": "reasoning", "name": "MBPP+", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500", "title": "MBPP+"}, {"benchmarks": [{"benchmark_id": "llm-stats-mbpp-base-version", "domain": "reasoning", "name": "MBPP ++ base version", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500", "title": "MBPP ++ base version"}, {"benchmarks": [{"benchmark_id": "llm-stats-mbpp-evalplus-base", "domain": "reasoning", "name": "MBPP EvalPlus (base)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500", "title": "MBPP EvalPlus (base)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mbpp-evalplus", "domain": "reasoning", "name": "MBPP EvalPlus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500", "title": "MBPP EvalPlus"}, {"benchmarks": [{"benchmark_id": "llm-stats-mbpp-pass-1", "domain": "reasoning", "name": "MBPP pass@1", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500", "title": "MBPP pass@1"}, {"benchmarks": [{"benchmark_id": "llm-stats-mbpp-plus", "domain": "reasoning", "name": "MBPP Plus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500", "title": "MBPP Plus"}, {"benchmarks": [{"benchmark_id": "llm-stats-mbpp", "domain": "reasoning", "name": "MBPP", "released": "2021-08-16", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500", "title": "MBPP"}, {"benchmarks": [{"benchmark_id": "llm-stats-mcp-atlas", "domain": "reasoning", "name": "MCP Atlas", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500", "title": "MCP Atlas"}, {"benchmarks": [{"benchmark_id": "llm-stats-mcp-mark", "domain": "agents", "name": "MCP-Mark", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500", "title": "MCP-Mark"}, {"benchmarks": [{"benchmark_id": "llm-stats-mcp-universe", "domain": "agents", "name": "MCP-Universe", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500", "title": "MCP-Universe"}, {"benchmarks": [{"benchmark_id": "llm-stats-measurebench", "domain": "multimodal", "name": "MeasureBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500", "title": "MeasureBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-medchembench", "domain": "reasoning", "name": "MedChemBench (Internal)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500", "title": "MedChemBench (Internal)"}, {"benchmarks": [{"benchmark_id": "llm-stats-medxpertqa-mm", "domain": "medical", "name": "MedXpertQA-MM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500", "title": "MedXpertQA-MM"}, {"benchmarks": [{"benchmark_id": "llm-stats-medxpertqa", "domain": "multimodal", "name": "MedXpertQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500", "title": "MedXpertQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-mega-mlqa", "domain": "reasoning", "name": "MEGA MLQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500", "title": "MEGA MLQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-mega-tydi-qa", "domain": "reasoning", "name": "MEGA TyDi QA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500", "title": "MEGA TyDi QA"}, {"benchmarks": [{"benchmark_id": "llm-stats-mega-udpos", "domain": "language", "name": "MEGA UDPOS", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500", "title": "MEGA UDPOS"}, {"benchmarks": [{"benchmark_id": "llm-stats-mega-xcopa", "domain": "reasoning", "name": "MEGA XCOPA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500", "title": "MEGA XCOPA"}, {"benchmarks": [{"benchmark_id": "llm-stats-mega-xstorycloze", "domain": "reasoning", "name": "MEGA XStoryCloze", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500", "title": "MEGA XStoryCloze"}, {"benchmarks": [{"benchmark_id": "llm-stats-meld", "domain": "multimodal", "name": "Meld", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500", "title": "Meld"}, {"benchmarks": [{"benchmark_id": "llm-stats-meta-internal-coding-bench", "domain": "agents", "name": "Meta Internal Coding Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500", "title": "Meta Internal Coding Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mewc", "domain": "reasoning", "name": "MEWC", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500", "title": "MEWC"}, {"benchmarks": [{"benchmark_id": "llm-stats-mgsm", "domain": "math", "name": "MGSM", "released": "2022-10-06", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500", "title": "MGSM"}, {"benchmarks": [{"benchmark_id": "llm-stats-miabench", "domain": "multimodal", "name": "MIABench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500", "title": "MIABench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mimic-cxr", "domain": "multimodal", "name": "MIMIC CXR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500", "title": "MIMIC CXR"}, {"benchmarks": [{"benchmark_id": "llm-stats-mimo-coding-bench", "domain": "agents", "name": "MiMo Coding Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500", "title": "MiMo Coding Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-minerva", "domain": "multimodal", "name": "Minerva", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500", "title": "Minerva"}, {"benchmarks": [{"benchmark_id": "llm-stats-mle-bench-lite", "domain": "agents", "name": "MLE-Bench Lite", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500", "title": "MLE-Bench Lite"}, {"benchmarks": [{"benchmark_id": "llm-stats-mle-bench", "domain": "agents", "name": "MLE-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500", "title": "MLE-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mls-bench-lite", "domain": "reasoning", "name": "MLS-Bench Lite", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500", "title": "MLS-Bench Lite"}, {"benchmarks": [{"benchmark_id": "llm-stats-mlvu-m", "domain": "general", "name": "MLVU-M", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500", "title": "MLVU-M"}, {"benchmarks": [{"benchmark_id": "llm-stats-mlvu", "domain": "long_context", "name": "MLVU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500", "title": "MLVU"}, {"benchmarks": [{"benchmark_id": "llm-stats-mm-browsercomp", "domain": "multimodal", "name": "MM-BrowserComp", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500", "title": "MM-BrowserComp"}, {"benchmarks": [{"benchmark_id": "llm-stats-mm-clawbench", "domain": "agents", "name": "MM-ClawBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500", "title": "MM-ClawBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mm-if-eval", "domain": "multimodal", "name": "MM IF-Eval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500", "title": "MM IF-Eval"}, {"benchmarks": [{"benchmark_id": "llm-stats-mm-mind2web", "domain": "multimodal", "name": "MM-Mind2Web", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500", "title": "MM-Mind2Web"}, {"benchmarks": [{"benchmark_id": "llm-stats-mm-mt-bench", "domain": "multimodal", "name": "MM-MT-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500", "title": "MM-MT-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmau-music", "domain": "multimodal", "name": "MMAU Music", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500", "title": "MMAU Music"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmau-sound", "domain": "multimodal", "name": "MMAU Sound", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500", "title": "MMAU Sound"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmau-speech", "domain": "multimodal", "name": "MMAU Speech", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500", "title": "MMAU Speech"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmau", "domain": "multimodal", "name": "MMAU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500", "title": "MMAU"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmbc", "domain": "multimodal", "name": "MMBC", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500", "title": "MMBC"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmbench-v1-1", "domain": "multimodal", "name": "MMBench-V1.1", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500", "title": "MMBench-V1.1"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmbench-video", "domain": "multimodal", "name": "MMBench-Video", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500", "title": "MMBench-Video"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmbench", "domain": "multimodal", "name": "MMBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500", "title": "MMBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mme-realworld", "domain": "multimodal", "name": "MME-RealWorld", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500", "title": "MME-RealWorld"}, {"benchmarks": [{"benchmark_id": "llm-stats-mme", "domain": "multimodal", "name": "MME", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500", "title": "MME"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlongbench-128k", "domain": "long_context", "name": "MMLongBench-128K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500", "title": "MMLongBench-128K"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlongbench-doc", "domain": "long_context", "name": "MMLongBench-Doc", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500", "title": "MMLongBench-Doc"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-cot", "domain": "legal", "name": "MMLU (CoT)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500", "title": "MMLU (CoT)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-base", "domain": "legal", "name": "MMLU-Base", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500", "title": "MMLU-Base"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-chat", "domain": "legal", "name": "MMLU Chat", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500", "title": "MMLU Chat"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-french", "domain": "legal", "name": "MMLU French", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500", "title": "MMLU French"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-pro", "domain": "legal", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500", "title": "MMLU-Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-prox", "domain": "legal", "name": "MMLU-ProX", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500", "title": "MMLU-ProX"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-redux-2-0", "domain": "math", "name": "MMLU-redux-2.0", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500", "title": "MMLU-redux-2.0"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-redux", "domain": "math", "name": "MMLU-Redux", "released": "2024-06-06", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500", "title": "MMLU-Redux"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu-stem", "domain": "math", "name": "MMLU-STEM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500", "title": "MMLU-STEM"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmlu", "domain": "legal", "name": "MMLU", "released": "2020-09-07", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500", "title": "MMLU"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmmlu", "domain": "math", "name": "MMMLU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500", "title": "MMMLU"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmmu-val", "domain": "multimodal", "name": "MMMU (val)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500", "title": "MMMU (val)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmmu-validation", "domain": "multimodal", "name": "MMMU (validation)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500", "title": "MMMU (validation)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmmu-pro-with-tools", "domain": "multimodal", "name": "MMMU-Pro (with tools)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500", "title": "MMMU-Pro (with tools)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmmu-pro", "domain": "multimodal", "name": "MMMU-Pro", "released": "2024-09-04", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500", "title": "MMMU-Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500", "title": "MMMU"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmmuval", "domain": "multimodal", "name": "MMMUval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500", "title": "MMMUval"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmsearch-plus", "domain": "multimodal", "name": "MMSearch-Plus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500", "title": "MMSearch-Plus"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmsearch", "domain": "multimodal", "name": "MMSearch", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500", "title": "MMSearch"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmsibench", "domain": "multimodal", "name": "MMSIBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500", "title": "MMSIBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmstar", "domain": "multimodal", "name": "MMStar", "released": "2024-04-09", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500", "title": "MMStar"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmt-bench", "domain": "multimodal", "name": "MMT-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500", "title": "MMT-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmvet", "domain": "math", "name": "MMVet", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500", "title": "MMVet"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmvetgpt4turbo", "domain": "math", "name": "MMVetGPT4Turbo", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500", "title": "MMVetGPT4Turbo"}, {"benchmarks": [{"benchmark_id": "llm-stats-mmvu", "domain": "multimodal", "name": "MMVU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500", "title": "MMVU"}, {"benchmarks": [{"benchmark_id": "llm-stats-mobileminiwob-sr", "domain": "multimodal", "name": "MobileMiniWob++_SR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500", "title": "MobileMiniWob++_SR"}, {"benchmarks": [{"benchmark_id": "llm-stats-mobileworld", "domain": "multimodal", "name": "MobileWorld", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500", "title": "MobileWorld"}, {"benchmarks": [{"benchmark_id": "llm-stats-motionbench", "domain": "multimodal", "name": "MotionBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500", "title": "MotionBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-128k-2-needle", "domain": "long_context", "name": "MRCR 128K (2-needle)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500", "title": "MRCR 128K (2-needle)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-128k-4-needle", "domain": "long_context", "name": "MRCR 128K (4-needle)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500", "title": "MRCR 128K (4-needle)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-128k-8-needle", "domain": "long_context", "name": "MRCR 128K (8-needle)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500", "title": "MRCR 128K (8-needle)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-1m-pointwise", "domain": "long_context", "name": "MRCR 1M (pointwise)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500", "title": "MRCR 1M (pointwise)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-1m", "domain": "long_context", "name": "MRCR 1M", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500", "title": "MRCR 1M"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-64k-2-needle", "domain": "long_context", "name": "MRCR 64K (2-needle)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500", "title": "MRCR 64K (2-needle)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-64k-4-needle", "domain": "long_context", "name": "MRCR 64K (4-needle)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500", "title": "MRCR 64K (4-needle)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-64k-8-needle", "domain": "long_context", "name": "MRCR 64K (8-needle)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500", "title": "MRCR 64K (8-needle)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-v2-8-needle", "domain": "long_context", "name": "MRCR v2 (8-needle)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500", "title": "MRCR v2 (8-needle)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-v2-8-needle-512k-1m", "domain": "long_context", "name": "MRCR v2 (8-needle, 512K-1M)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500", "title": "MRCR v2 (8-needle, 512K-1M)"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr-v2", "domain": "long_context", "name": "MRCR v2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500", "title": "MRCR v2"}, {"benchmarks": [{"benchmark_id": "llm-stats-mrcr", "domain": "long_context", "name": "MRCR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500", "title": "MRCR"}, {"benchmarks": [{"benchmark_id": "llm-stats-msqa", "domain": "reasoning", "name": "MSQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500", "title": "MSQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-mt-aime-2025", "domain": "math", "name": "MT-AIME 2025", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500", "title": "MT-AIME 2025"}, {"benchmarks": [{"benchmark_id": "llm-stats-mt-bench", "domain": "reasoning", "name": "MT-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500", "title": "MT-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-mtvqa", "domain": "multimodal", "name": "MTVQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500", "title": "MTVQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-muirbench", "domain": "multimodal", "name": "MuirBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500", "title": "MuirBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-multi-if", "domain": "reasoning", "name": "Multi-IF", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500", "title": "Multi-IF"}, {"benchmarks": [{"benchmark_id": "llm-stats-multi-swe-bench", "domain": "reasoning", "name": "Multi-SWE-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500", "title": "Multi-SWE-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-multichallenge", "domain": "reasoning", "name": "Multi-Challenge", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500", "title": "Multi-Challenge"}, {"benchmarks": [{"benchmark_id": "llm-stats-multilf", "domain": "general", "name": "MultiLF", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500", "title": "MultiLF"}, {"benchmarks": [{"benchmark_id": "llm-stats-multilingual-mgsm-cot", "domain": "math", "name": "Multilingual MGSM (CoT)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500", "title": "Multilingual MGSM (CoT)"}, {"benchmarks": [{"benchmark_id": "llm-stats-multilingual-mmlu", "domain": "reasoning", "name": "Multilingual MMLU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500", "title": "Multilingual MMLU"}, {"benchmarks": [{"benchmark_id": "llm-stats-multipl-e-humaneval", "domain": "language", "name": "Multipl-E HumanEval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500", "title": "Multipl-E HumanEval"}, {"benchmarks": [{"benchmark_id": "llm-stats-multipl-e-mbpp", "domain": "reasoning", "name": "Multipl-E MBPP", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500", "title": "Multipl-E MBPP"}, {"benchmarks": [{"benchmark_id": "llm-stats-multipl-e", "domain": "language", "name": "MultiPL-E", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500", "title": "MultiPL-E"}, {"benchmarks": [{"benchmark_id": "llm-stats-musiccaps", "domain": "multimodal", "name": "MusicCaps", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500", "title": "MusicCaps"}, {"benchmarks": [{"benchmark_id": "llm-stats-musr", "domain": "reasoning", "name": "MuSR", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500", "title": "MuSR"}, {"benchmarks": [{"benchmark_id": "llm-stats-mvbench", "domain": "multimodal", "name": "MVBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500", "title": "MVBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-nanogpt", "domain": "agents", "name": "NanoGPT", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500", "title": "NanoGPT"}, {"benchmarks": [{"benchmark_id": "llm-stats-natural-questions", "domain": "reasoning", "name": "Natural Questions", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500", "title": "Natural Questions"}, {"benchmarks": [{"benchmark_id": "llm-stats-natural2code", "domain": "reasoning", "name": "Natural2Code", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500", "title": "Natural2Code"}, {"benchmarks": [{"benchmark_id": "llm-stats-nexus", "domain": "general", "name": "Nexus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500", "title": "Nexus"}, {"benchmarks": [{"benchmark_id": "llm-stats-nih-multi-needle", "domain": "long_context", "name": "NIH/Multi-needle", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500", "title": "NIH/Multi-needle"}, {"benchmarks": [{"benchmark_id": "llm-stats-nl2repo", "domain": "agents", "name": "NL2Repo", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500", "title": "NL2Repo"}, {"benchmarks": [{"benchmark_id": "llm-stats-nmos", "domain": "general", "name": "NMOS", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500", "title": "NMOS"}, {"benchmarks": [{"benchmark_id": "llm-stats-nolima-128k", "domain": "long_context", "name": "NoLiMa 128K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500", "title": "NoLiMa 128K"}, {"benchmarks": [{"benchmark_id": "llm-stats-nolima-32k", "domain": "long_context", "name": "NoLiMa 32K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500", "title": "NoLiMa 32K"}, {"benchmarks": [{"benchmark_id": "llm-stats-nolima-64k", "domain": "long_context", "name": "NoLiMa 64K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500", "title": "NoLiMa 64K"}, {"benchmarks": [{"benchmark_id": "llm-stats-nova-63", "domain": "general", "name": "NOVA-63", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500", "title": "NOVA-63"}, {"benchmarks": [{"benchmark_id": "llm-stats-nq", "domain": "reasoning", "name": "NQ", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500", "title": "NQ"}, {"benchmarks": [{"benchmark_id": "llm-stats-nuscene", "domain": "multimodal", "name": "Nuscene", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500", "title": "Nuscene"}, {"benchmarks": [{"benchmark_id": "llm-stats-objectron", "domain": "3d", "name": "Objectron", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500", "title": "Objectron"}, {"benchmarks": [{"benchmark_id": "llm-stats-ocrbench-v2-en", "domain": "image_to_text", "name": "OCRBench-V2 (en)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500", "title": "OCRBench-V2 (en)"}, {"benchmarks": [{"benchmark_id": "llm-stats-ocrbench-v2-zh", "domain": "image_to_text", "name": "OCRBench-V2 (zh)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500", "title": "OCRBench-V2 (zh)"}, {"benchmarks": [{"benchmark_id": "llm-stats-ocrbench-v2", "domain": "image_to_text", "name": "OCRBench_V2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500", "title": "OCRBench_V2"}, {"benchmarks": [{"benchmark_id": "llm-stats-ocrbench", "domain": "image_to_text", "name": "OCRBench", "released": "2024-01-17", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500", "title": "OCRBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-octocodingbench", "domain": "code", "name": "OctoCodingBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500", "title": "OctoCodingBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-odinw", "domain": "vision", "name": "ODinW", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500", "title": "ODinW"}, {"benchmarks": [{"benchmark_id": "llm-stats-officeqa-pro", "domain": "reasoning", "name": "OfficeQA Pro", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500", "title": "OfficeQA Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-ojbench-cpp", "domain": "reasoning", "name": "OJBench (C++)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500", "title": "OJBench (C++)"}, {"benchmarks": [{"benchmark_id": "llm-stats-ojbench", "domain": "reasoning", "name": "OJBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500", "title": "OJBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-olympiadbench", "domain": "math", "name": "OlympiadBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500", "title": "OlympiadBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-omnibench-music", "domain": "multimodal", "name": "OmniBench Music", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500", "title": "OmniBench Music"}, {"benchmarks": [{"benchmark_id": "llm-stats-omnibench", "domain": "multimodal", "name": "OmniBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500", "title": "OmniBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-omnidocbench-1-5", "domain": "multimodal", "name": "OmniDocBench 1.5", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500", "title": "OmniDocBench 1.5"}, {"benchmarks": [{"benchmark_id": "llm-stats-omnidocbench", "domain": "multimodal", "name": "OmniDocBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500", "title": "OmniDocBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-omnigaia", "domain": "multimodal", "name": "OmniGAIA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500", "title": "OmniGAIA"}, {"benchmarks": [{"benchmark_id": "llm-stats-omnimath", "domain": "math", "name": "OmniMath", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500", "title": "OmniMath"}, {"benchmarks": [{"benchmark_id": "llm-stats-omniscience-non-hallucination-rate", "domain": "reasoning", "name": "OmniScience (non-hallucination rate)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500", "title": "OmniScience (non-hallucination rate)"}, {"benchmarks": [{"benchmark_id": "llm-stats-omniscience", "domain": "reasoning", "name": "OmniScience", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500", "title": "OmniScience"}, {"benchmarks": [{"benchmark_id": "llm-stats-onemillion-bench", "domain": "long_context", "name": "OneMillion Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500", "title": "OneMillion Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-open-rewrite", "domain": "language", "name": "Open-rewrite", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500", "title": "Open-rewrite"}, {"benchmarks": [{"benchmark_id": "llm-stats-openai-connectors", "domain": "agents", "name": "Connectors", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500", "title": "Connectors"}, {"benchmarks": [{"benchmark_id": "llm-stats-openai-mmlu", "domain": "legal", "name": "OpenAI MMLU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500", "title": "OpenAI MMLU"}, {"benchmarks": [{"benchmark_id": "llm-stats-openai-mrcr-2-needle-128k", "domain": "long_context", "name": "OpenAI-MRCR: 2 needle 128k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500", "title": "OpenAI-MRCR: 2 needle 128k"}, {"benchmarks": [{"benchmark_id": "llm-stats-openai-mrcr-2-needle-1m", "domain": "long_context", "name": "OpenAI-MRCR: 2 needle 1M", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500", "title": "OpenAI-MRCR: 2 needle 1M"}, {"benchmarks": [{"benchmark_id": "llm-stats-openai-mrcr-2-needle-256k", "domain": "long_context", "name": "OpenAI-MRCR: 2 needle 256k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500", "title": "OpenAI-MRCR: 2 needle 256k"}, {"benchmarks": [{"benchmark_id": "llm-stats-openai-search-function-calling", "domain": "agents", "name": "Search and Function-Calling", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500", "title": "Search and Function-Calling"}, {"benchmarks": [{"benchmark_id": "llm-stats-openbookqa", "domain": "reasoning", "name": "OpenBookQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500", "title": "OpenBookQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-openrca", "domain": "reasoning", "name": "OpenRCA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500", "title": "OpenRCA"}, {"benchmarks": [{"benchmark_id": "llm-stats-osworld-2-0", "domain": "multimodal", "name": "OSWorld 2.0", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500", "title": "OSWorld 2.0"}, {"benchmarks": [{"benchmark_id": "llm-stats-osworld-extended", "domain": "multimodal", "name": "OSWorld Extended", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500", "title": "OSWorld Extended"}, {"benchmarks": [{"benchmark_id": "llm-stats-osworld-g", "domain": "multimodal", "name": "OSWorld-G", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500", "title": "OSWorld-G"}, {"benchmarks": [{"benchmark_id": "llm-stats-osworld-screenshot-only", "domain": "multimodal", "name": "OSWorld Screenshot-only", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500", "title": "OSWorld Screenshot-only"}, {"benchmarks": [{"benchmark_id": "llm-stats-osworld-verified", "domain": "multimodal", "name": "OSWorld-Verified", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500", "title": "OSWorld-Verified"}, {"benchmarks": [{"benchmark_id": "llm-stats-osworld", "domain": "multimodal", "name": "OSWorld", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500", "title": "OSWorld"}, {"benchmarks": [{"benchmark_id": "llm-stats-ovbench", "domain": "multimodal", "name": "OVBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500", "title": "OVBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-ovobench", "domain": "multimodal", "name": "OVOBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500", "title": "OVOBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-paperbench", "domain": "reasoning", "name": "PaperBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500", "title": "PaperBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-pathmcqa", "domain": "multimodal", "name": "PathMCQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500", "title": "PathMCQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-perceptionbench", "domain": "multimodal", "name": "PerceptionBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500", "title": "PerceptionBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-perceptiontest", "domain": "multimodal", "name": "PerceptionTest", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500", "title": "PerceptionTest"}, {"benchmarks": [{"benchmark_id": "llm-stats-phibench", "domain": "math", "name": "PhiBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500", "title": "PhiBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-phybench", "domain": "physics", "name": "PHYBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500", "title": "PHYBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-physicsfinals", "domain": "math", "name": "PhysicsFinals", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500", "title": "PhysicsFinals"}, {"benchmarks": [{"benchmark_id": "llm-stats-pinchbench", "domain": "agents", "name": "PinchBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500", "title": "PinchBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-piqa", "domain": "physics", "name": "PIQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500", "title": "PIQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-plawbench", "domain": "legal", "name": "PLawBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500", "title": "PLawBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-pmc-vqa", "domain": "multimodal", "name": "PMC-VQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500", "title": "PMC-VQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-pointgrounding", "domain": "multimodal", "name": "PointGrounding", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500", "title": "PointGrounding"}, {"benchmarks": [{"benchmark_id": "llm-stats-polymath-en", "domain": "math", "name": "PolyMath-en", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500", "title": "PolyMath-en"}, {"benchmarks": [{"benchmark_id": "llm-stats-polymath", "domain": "math", "name": "PolyMATH", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500", "title": "PolyMATH"}, {"benchmarks": [{"benchmark_id": "llm-stats-pope", "domain": "multimodal", "name": "POPE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500", "title": "POPE"}, {"benchmarks": [{"benchmark_id": "llm-stats-popqa", "domain": "reasoning", "name": "PopQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500", "title": "PopQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-posttrainbench-lite", "domain": "reasoning", "name": "PostTrainBench Lite", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500", "title": "PostTrainBench Lite"}, {"benchmarks": [{"benchmark_id": "llm-stats-posttrainbench", "domain": "reasoning", "name": "PostTrainBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500", "title": "PostTrainBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-prbench-finance", "domain": "reasoning", "name": "PRBench-Finance", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500", "title": "PRBench-Finance"}, {"benchmarks": [{"benchmark_id": "llm-stats-prbench-legal", "domain": "legal", "name": "PRBench-Legal", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500", "title": "PRBench-Legal"}, {"benchmarks": [{"benchmark_id": "llm-stats-presentbench", "domain": "reasoning", "name": "PresentBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500", "title": "PresentBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-profbench", "domain": "reasoning", "name": "ProfBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500", "title": "ProfBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-program-bench", "domain": "agents", "name": "Program Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500", "title": "Program Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-protocolqa", "domain": "safety", "name": "ProtocolQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500", "title": "ProtocolQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-qasper", "domain": "long_context", "name": "Qasper", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500", "title": "Qasper"}, {"benchmarks": [{"benchmark_id": "llm-stats-qmsum", "domain": "long_context", "name": "QMSum", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500", "title": "QMSum"}, {"benchmarks": [{"benchmark_id": "llm-stats-qvhighlights", "domain": "multimodal", "name": "QVHighlights", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500", "title": "QVHighlights"}, {"benchmarks": [{"benchmark_id": "llm-stats-qwen-qoder-bench", "domain": "agents", "name": "QwenQoderBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500", "title": "QwenQoderBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-qwen-react-bench", "domain": "multimodal", "name": "QwenReactBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500", "title": "QwenReactBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-qwen-svg", "domain": "multimodal", "name": "QwenSVG", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500", "title": "QwenSVG"}, {"benchmarks": [{"benchmark_id": "llm-stats-qwen-swe-bench", "domain": "agents", "name": "QwenSWEBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500", "title": "QwenSWEBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-qwenclawbench", "domain": "agents", "name": "QwenClawBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500", "title": "QwenClawBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-qwenwebbench", "domain": "multimodal", "name": "QwenWebBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500", "title": "QwenWebBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-qwenworldbench", "domain": "reasoning", "name": "QwenWorldBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500", "title": "QwenWorldBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-realkie-fcc", "domain": "multimodal", "name": "RealKIE-FCC", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500", "title": "RealKIE-FCC"}, {"benchmarks": [{"benchmark_id": "llm-stats-realworldqa", "domain": "spatial_reasoning", "name": "RealWorldQA", "released": "2024-04-12", "url": "https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500", "title": "RealWorldQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-recreationbench", "domain": "multimodal", "name": "RecreationBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500", "title": "RecreationBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-refcoco-avg", "domain": "spatial_reasoning", "name": "RefCOCO-avg", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500", "title": "RefCOCO-avg"}, {"benchmarks": [{"benchmark_id": "llm-stats-refcocog", "domain": "multimodal", "name": "RefCOCOg", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500", "title": "RefCOCOg"}, {"benchmarks": [{"benchmark_id": "llm-stats-refspatialbench", "domain": "spatial_reasoning", "name": "RefSpatialBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500", "title": "RefSpatialBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-repo-env", "domain": "agents", "name": "Repo Env", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500", "title": "Repo Env"}, {"benchmarks": [{"benchmark_id": "llm-stats-repobench", "domain": "reasoning", "name": "RepoBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500", "title": "RepoBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-repoqa", "domain": "long_context", "name": "RepoQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500", "title": "RepoQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-researchclawbench", "domain": "research", "name": "ResearchClawBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500", "title": "ResearchClawBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-robospatialhome", "domain": "robotics", "name": "RoboSpatialHome", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500", "title": "RoboSpatialHome"}, {"benchmarks": [{"benchmark_id": "llm-stats-robust-if", "domain": "reasoning", "name": "Robust IF", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500", "title": "Robust IF"}, {"benchmarks": [{"benchmark_id": "llm-stats-rsi-index", "domain": "reasoning", "name": "RSI Index", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500", "title": "RSI Index"}, {"benchmarks": [{"benchmark_id": "llm-stats-ruler-1000k", "domain": "long_context", "name": "RULER 1000K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500", "title": "RULER 1000K"}, {"benchmarks": [{"benchmark_id": "llm-stats-ruler-128k", "domain": "long_context", "name": "RULER 128k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500", "title": "RULER 128k"}, {"benchmarks": [{"benchmark_id": "llm-stats-ruler-2048k", "domain": "long_context", "name": "RULER 2048K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500", "title": "RULER 2048K"}, {"benchmarks": [{"benchmark_id": "llm-stats-ruler-512k", "domain": "long_context", "name": "RULER 512K", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500", "title": "RULER 512K"}, {"benchmarks": [{"benchmark_id": "llm-stats-ruler-64k", "domain": "long_context", "name": "RULER 64k", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500", "title": "RULER 64k"}, {"benchmarks": [{"benchmark_id": "llm-stats-ruler", "domain": "long_context", "name": "RULER", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500", "title": "RULER"}, {"benchmarks": [{"benchmark_id": "llm-stats-sat-math", "domain": "math", "name": "SAT Math", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500", "title": "SAT Math"}, {"benchmarks": [{"benchmark_id": "llm-stats-scicode", "domain": "math", "name": "SciCode", "released": "2024-07-18", "url": "https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500", "title": "SciCode"}, {"benchmarks": [{"benchmark_id": "llm-stats-scienceqa-visual", "domain": "multimodal", "name": "ScienceQA Visual", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500", "title": "ScienceQA Visual"}, {"benchmarks": [{"benchmark_id": "llm-stats-scienceqa", "domain": "math", "name": "ScienceQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500", "title": "ScienceQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-screenspot-pro", "domain": "multimodal", "name": "ScreenSpot Pro", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500", "title": "ScreenSpot Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-screenspot", "domain": "multimodal", "name": "ScreenSpot", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500", "title": "ScreenSpot"}, {"benchmarks": [{"benchmark_id": "llm-stats-seal-0", "domain": "reasoning", "name": "Seal-0", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500", "title": "Seal-0"}, {"benchmarks": [{"benchmark_id": "llm-stats-sec-bench-pro", "domain": "safety", "name": "SEC-bench Pro", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500", "title": "SEC-bench Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-seccodebench", "domain": "code", "name": "SecCodeBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500", "title": "SecCodeBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-seedclawbench", "domain": "agents", "name": "SeedClawBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500", "title": "SeedClawBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-sifo-multiturn", "domain": "structured_output", "name": "SIFO-Multiturn", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500", "title": "SIFO-Multiturn"}, {"benchmarks": [{"benchmark_id": "llm-stats-sifo", "domain": "structured_output", "name": "SIFO", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500", "title": "SIFO"}, {"benchmarks": [{"benchmark_id": "llm-stats-simpleqa-verified", "domain": "reasoning", "name": "SimpleQA Verified", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500", "title": "SimpleQA Verified"}, {"benchmarks": [{"benchmark_id": "llm-stats-simpleqa", "domain": "reasoning", "name": "SimpleQA", "released": "2024-10-30", "url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500", "title": "SimpleQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-simplevqa", "domain": "multimodal", "name": "SimpleVQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500", "title": "SimpleVQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-siren-agentdojo-attack-success", "domain": "safety", "name": "Siren AgentDojo Attack Success Rate", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500", "title": "Siren AgentDojo Attack Success Rate"}, {"benchmarks": [{"benchmark_id": "llm-stats-siren-agentdojo-utility", "domain": "safety", "name": "Siren AgentDojo Utility", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500", "title": "Siren AgentDojo Utility"}, {"benchmarks": [{"benchmark_id": "llm-stats-skillsbench", "domain": "agents", "name": "SkillsBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500", "title": "SkillsBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-slakevqa", "domain": "multimodal", "name": "SlakeVQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500", "title": "SlakeVQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-social-iqa", "domain": "psychology", "name": "Social IQa", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500", "title": "Social IQa"}, {"benchmarks": [{"benchmark_id": "llm-stats-spider", "domain": "reasoning", "name": "Spider", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500", "title": "Spider"}, {"benchmarks": [{"benchmark_id": "llm-stats-spreadsheetbench-2", "domain": "productivity", "name": "SpreadsheetBench 2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500", "title": "SpreadsheetBench 2"}, {"benchmarks": [{"benchmark_id": "llm-stats-spreadsheetbench-v1", "domain": "productivity", "name": "SpreadSheetBench-v1", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500", "title": "SpreadSheetBench-v1"}, {"benchmarks": [{"benchmark_id": "llm-stats-squality", "domain": "long_context", "name": "SQuALITY", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500", "title": "SQuALITY"}, {"benchmarks": [{"benchmark_id": "llm-stats-stem", "domain": "math", "name": "STEM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500", "title": "STEM"}, {"benchmarks": [{"benchmark_id": "llm-stats-summscreenfd", "domain": "long_context", "name": "SummScreenFD", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500", "title": "SummScreenFD"}, {"benchmarks": [{"benchmark_id": "llm-stats-sunrgbd", "domain": "spatial_reasoning", "name": "SUNRGBD", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500", "title": "SUNRGBD"}, {"benchmarks": [{"benchmark_id": "llm-stats-superchem", "domain": "reasoning", "name": "SuperChem", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500", "title": "SuperChem"}, {"benchmarks": [{"benchmark_id": "llm-stats-superglue", "domain": "reasoning", "name": "SuperGLUE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500", "title": "SuperGLUE"}, {"benchmarks": [{"benchmark_id": "llm-stats-supergpqa", "domain": "legal", "name": "SuperGPQA", "released": "2025-02-20", "url": "https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500", "title": "SuperGPQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-surds", "domain": "multimodal", "name": "SURDS", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500", "title": "SURDS"}, {"benchmarks": [{"benchmark_id": "llm-stats-svg-bench", "domain": "multimodal", "name": "SVG-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500", "title": "SVG-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-atlas-codebase-qna", "domain": "agents", "name": "SWE Atlas - Codebase QnA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500", "title": "SWE Atlas - Codebase QnA"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-atlas-test-writing", "domain": "agents", "name": "SWE Atlas - Test Writing", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500", "title": "SWE Atlas - Test Writing"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-atlas", "domain": "agents", "name": "SWE-Atlas", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500", "title": "SWE-Atlas"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-bench-multilingual", "domain": "reasoning", "name": "SWE-bench Multilingual", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500", "title": "SWE-bench Multilingual"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-bench-multimodal", "domain": "multimodal", "name": "SWE-Bench Multimodal", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500", "title": "SWE-Bench Multimodal"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-bench-pro", "domain": "reasoning", "name": "SWE-Bench Pro", "released": "2025-09-05", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500", "title": "SWE-Bench Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-bench-verified-agentic-coding", "domain": "reasoning", "name": "SWE-bench Verified (Agentic Coding)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500", "title": "SWE-bench Verified (Agentic Coding)"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-bench-verified-agentless", "domain": "reasoning", "name": "SWE-bench Verified (Agentless)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500", "title": "SWE-bench Verified (Agentless)"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-bench-verified-multiple-attempts", "domain": "reasoning", "name": "SWE-bench Verified (Multiple Attempts)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500", "title": "SWE-bench Verified (Multiple Attempts)"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-bench-verified", "domain": "reasoning", "name": "SWE-Bench Verified", "released": "2024-08-13", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500", "title": "SWE-Bench Verified"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-fficiency", "domain": "agents", "name": "SWE-fficiency", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500", "title": "SWE-fficiency"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-lancer-ic-diamond-subset", "domain": "reasoning", "name": "SWE-Lancer (IC-Diamond subset)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500", "title": "SWE-Lancer (IC-Diamond subset)"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-lancer", "domain": "reasoning", "name": "SWE-Lancer", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500", "title": "SWE-Lancer"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-marathon", "domain": "agents", "name": "SWE-Marathon", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500", "title": "SWE-Marathon"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-mm", "domain": "multimodal", "name": "SWE-MM", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500", "title": "SWE-MM"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-perf", "domain": "code", "name": "SWE-Perf", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500", "title": "SWE-Perf"}, {"benchmarks": [{"benchmark_id": "llm-stats-swe-review", "domain": "code", "name": "SWE-Review", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500", "title": "SWE-Review"}, {"benchmarks": [{"benchmark_id": "llm-stats-swt-bench", "domain": "code", "name": "SWT-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500", "title": "SWT-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-t2-bench", "domain": "reasoning", "name": "t2-bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500", "title": "t2-bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau-bench-airline", "domain": "reasoning", "name": "TAU-bench Airline", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500", "title": "TAU-bench Airline"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau-bench-retail", "domain": "reasoning", "name": "TAU-bench Retail", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500", "title": "TAU-bench Retail"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau-bench", "domain": "reasoning", "name": "Tau-bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500", "title": "Tau-bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau2-airline", "domain": "reasoning", "name": "Tau2 Airline", "released": "2025-06-09", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500", "title": "Tau2 Airline"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau2-retail", "domain": "reasoning", "name": "Tau2 Retail", "released": "2025-06-09", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500", "title": "Tau2 Retail"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau2-telecom", "domain": "reasoning", "name": "Tau2 Telecom", "released": "2025-06-09", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500", "title": "Tau2 Telecom"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau3-airline", "domain": "reasoning", "name": "Tau3 Airline", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500", "title": "Tau3 Airline"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau3-banking", "domain": "reasoning", "name": "Tau3 Banking", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500", "title": "Tau3 Banking"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau3-bench", "domain": "reasoning", "name": "TAU3-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500", "title": "TAU3-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau3-retail", "domain": "reasoning", "name": "Tau3 Retail", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500", "title": "Tau3 Retail"}, {"benchmarks": [{"benchmark_id": "llm-stats-tau3-telecom", "domain": "reasoning", "name": "Tau3 Telecom", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500", "title": "Tau3 Telecom"}, {"benchmarks": [{"benchmark_id": "llm-stats-tempcompass", "domain": "multimodal", "name": "TempCompass", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500", "title": "TempCompass"}, {"benchmarks": [{"benchmark_id": "llm-stats-terminal-bench-2-1", "domain": "reasoning", "name": "Terminal-Bench 2.1", "released": "2026-05-05", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500", "title": "Terminal-Bench 2.1"}, {"benchmarks": [{"benchmark_id": "llm-stats-terminal-bench-2", "domain": "reasoning", "name": "Terminal-Bench 2.0", "released": "2025-09-25", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500", "title": "Terminal-Bench 2.0"}, {"benchmarks": [{"benchmark_id": "llm-stats-terminal-bench-3-0", "domain": "reasoning", "name": "Terminal-Bench 3.0", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500", "title": "Terminal-Bench 3.0"}, {"benchmarks": [{"benchmark_id": "llm-stats-terminal-bench-hard", "domain": "reasoning", "name": "Terminal-Bench Hard", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500", "title": "Terminal-Bench Hard"}, {"benchmarks": [{"benchmark_id": "llm-stats-terminal-bench", "domain": "reasoning", "name": "Terminal-Bench", "released": "2025-01-17", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500", "title": "Terminal-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-terminus", "domain": "reasoning", "name": "Terminus", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500", "title": "Terminus"}, {"benchmarks": [{"benchmark_id": "llm-stats-textvqa", "domain": "multimodal", "name": "TextVQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500", "title": "TextVQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-theoremqa", "domain": "math", "name": "TheoremQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500", "title": "TheoremQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-tir-bench", "domain": "multimodal", "name": "TIR-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500", "title": "TIR-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-tldr9-test", "domain": "summarization", "name": "TLDR9+ (test)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500", "title": "TLDR9+ (test)"}, {"benchmarks": [{"benchmark_id": "llm-stats-tomato", "domain": "multimodal", "name": "TOMATO", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500", "title": "TOMATO"}, {"benchmarks": [{"benchmark_id": "llm-stats-toolathlon", "domain": "reasoning", "name": "Toolathlon", "released": "2025-10-29", "url": "https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500", "title": "Toolathlon"}, {"benchmarks": [{"benchmark_id": "llm-stats-trae-code-gen", "domain": "agents", "name": "Trae Code Gen", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500", "title": "Trae Code Gen"}, {"benchmarks": [{"benchmark_id": "llm-stats-trae-error-fix", "domain": "agents", "name": "Trae Error Fix", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500", "title": "Trae Error Fix"}, {"benchmarks": [{"benchmark_id": "llm-stats-translation-en-set1-comet22", "domain": "language", "name": "Translation en→Set1 COMET22", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500", "title": "Translation en→Set1 COMET22"}, {"benchmarks": [{"benchmark_id": "llm-stats-translation-en-set1-spbleu", "domain": "language", "name": "Translation en→Set1 spBleu", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500", "title": "Translation en→Set1 spBleu"}, {"benchmarks": [{"benchmark_id": "llm-stats-translation-set1-en-comet22", "domain": "language", "name": "Translation Set1→en COMET22", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500", "title": "Translation Set1→en COMET22"}, {"benchmarks": [{"benchmark_id": "llm-stats-translation-set1-en-spbleu", "domain": "language", "name": "Translation Set1→en spBleu", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500", "title": "Translation Set1→en spBleu"}, {"benchmarks": [{"benchmark_id": "llm-stats-treebench", "domain": "multimodal", "name": "TreeBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500", "title": "TreeBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-triviaqa", "domain": "reasoning", "name": "TriviaQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500", "title": "TriviaQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-truthfulqa", "domain": "legal", "name": "TruthfulQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500", "title": "TruthfulQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-tvbench", "domain": "multimodal", "name": "TVBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500", "title": "TVBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-tydiqa", "domain": "reasoning", "name": "TydiQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500", "title": "TydiQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-uniform-bar-exam", "domain": "legal", "name": "Uniform Bar Exam", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500", "title": "Uniform Bar Exam"}, {"benchmarks": [{"benchmark_id": "llm-stats-usamo-2026", "domain": "math", "name": "USAMO 2026", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500", "title": "USAMO 2026"}, {"benchmarks": [{"benchmark_id": "llm-stats-usamo25", "domain": "math", "name": "USAMO25", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500", "title": "USAMO25"}, {"benchmarks": [{"benchmark_id": "llm-stats-v-star", "domain": "multimodal", "name": "V*", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500", "title": "V*"}, {"benchmarks": [{"benchmark_id": "llm-stats-vatex", "domain": "multimodal", "name": "VATEX", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500", "title": "VATEX"}, {"benchmarks": [{"benchmark_id": "llm-stats-vcr-en-easy", "domain": "reasoning", "name": "VCR_en_easy", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500", "title": "VCR_en_easy"}, {"benchmarks": [{"benchmark_id": "llm-stats-vct", "domain": "safety", "name": "Virology Capabilities Test", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500", "title": "Virology Capabilities Test"}, {"benchmarks": [{"benchmark_id": "llm-stats-vending-bench-2", "domain": "reasoning", "name": "Vending-Bench 2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500", "title": "Vending-Bench 2"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-android", "domain": "code", "name": "VIBE Android", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500", "title": "VIBE Android"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-backend", "domain": "code", "name": "VIBE Backend", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500", "title": "VIBE Backend"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-eval", "domain": "multimodal", "name": "Vibe-Eval", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500", "title": "Vibe-Eval"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-ios", "domain": "code", "name": "VIBE iOS", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500", "title": "VIBE iOS"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-pro", "domain": "agents", "name": "VIBE-Pro", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500", "title": "VIBE-Pro"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-simulation", "domain": "code", "name": "VIBE Simulation", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500", "title": "VIBE Simulation"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-v2", "domain": "agents", "name": "VIBE-V2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500", "title": "VIBE-V2"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe-web", "domain": "code", "name": "VIBE Web", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500", "title": "VIBE Web"}, {"benchmarks": [{"benchmark_id": "llm-stats-vibe", "domain": "code", "name": "VIBE", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500", "title": "VIBE"}, {"benchmarks": [{"benchmark_id": "llm-stats-video-mme-long-no-subtitles", "domain": "multimodal", "name": "Video-MME (long, no subtitles)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500", "title": "Video-MME (long, no subtitles)"}, {"benchmarks": [{"benchmark_id": "llm-stats-video-mme", "domain": "multimodal", "name": "Video-MME", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500", "title": "Video-MME"}, {"benchmarks": [{"benchmark_id": "llm-stats-videoholmes", "domain": "multimodal", "name": "VideoHolmes", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500", "title": "VideoHolmes"}, {"benchmarks": [{"benchmark_id": "llm-stats-videomme-w-o-sub", "domain": "multimodal", "name": "VideoMME w/o sub.", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500", "title": "VideoMME w/o sub."}, {"benchmarks": [{"benchmark_id": "llm-stats-videomme-w-sub", "domain": "multimodal", "name": "VideoMME w sub.", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500", "title": "VideoMME w sub."}, {"benchmarks": [{"benchmark_id": "llm-stats-videommmu", "domain": "multimodal", "name": "VideoMMMU", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500", "title": "VideoMMMU"}, {"benchmarks": [{"benchmark_id": "llm-stats-videosimpleqa", "domain": "multimodal", "name": "VideoSimpleQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500", "title": "VideoSimpleQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-visfactor", "domain": "multimodal", "name": "VisFactor", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500", "title": "VisFactor"}, {"benchmarks": [{"benchmark_id": "llm-stats-vision2web", "domain": "multimodal", "name": "Vision2Web", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500", "title": "Vision2Web"}, {"benchmarks": [{"benchmark_id": "llm-stats-visualwebbench", "domain": "multimodal", "name": "VisualWebBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500", "title": "VisualWebBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-visulogic", "domain": "multimodal", "name": "VisuLogic", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500", "title": "VisuLogic"}, {"benchmarks": [{"benchmark_id": "llm-stats-vita-bench", "domain": "reasoning", "name": "VITA-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500", "title": "VITA-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-vladbench", "domain": "multimodal", "name": "VLADBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500", "title": "VLADBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-vlmsarebiased", "domain": "multimodal", "name": "VLMsAreBiased", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500", "title": "VLMsAreBiased"}, {"benchmarks": [{"benchmark_id": "llm-stats-vlmsareblind", "domain": "multimodal", "name": "VLMsAreBlind", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500", "title": "VLMsAreBlind"}, {"benchmarks": [{"benchmark_id": "llm-stats-vocalsound", "domain": "audio", "name": "VocalSound", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500", "title": "VocalSound"}, {"benchmarks": [{"benchmark_id": "llm-stats-voicebench-avg", "domain": "reasoning", "name": "VoiceBench Avg", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500", "title": "VoiceBench Avg"}, {"benchmarks": [{"benchmark_id": "llm-stats-vqa-rad", "domain": "multimodal", "name": "VQA-Rad", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500", "title": "VQA-Rad"}, {"benchmarks": [{"benchmark_id": "llm-stats-vqav2-test", "domain": "multimodal", "name": "VQAv2 (test)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500", "title": "VQAv2 (test)"}, {"benchmarks": [{"benchmark_id": "llm-stats-vqav2-val", "domain": "multimodal", "name": "VQAv2 (val)", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500", "title": "VQAv2 (val)"}, {"benchmarks": [{"benchmark_id": "llm-stats-vqav2", "domain": "multimodal", "name": "VQAv2", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500", "title": "VQAv2"}, {"benchmarks": [{"benchmark_id": "llm-stats-we-math", "domain": "math", "name": "We-Math", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500", "title": "We-Math"}, {"benchmarks": [{"benchmark_id": "llm-stats-web-bench", "domain": "agents", "name": "Web Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500", "title": "Web Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-webarena-verified", "domain": "multimodal", "name": "WebArena-Verified", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500", "title": "WebArena-Verified"}, {"benchmarks": [{"benchmark_id": "llm-stats-webdev-arena", "domain": "reasoning", "name": "WebDev Arena", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500", "title": "WebDev Arena"}, {"benchmarks": [{"benchmark_id": "llm-stats-webvoyager", "domain": "agents", "name": "WebVoyager", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500", "title": "WebVoyager"}, {"benchmarks": [{"benchmark_id": "llm-stats-widesearch", "domain": "reasoning", "name": "WideSearch", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500", "title": "WideSearch"}, {"benchmarks": [{"benchmark_id": "llm-stats-wild-bench", "domain": "reasoning", "name": "Wild Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500", "title": "Wild Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-wildclawbench", "domain": "agents", "name": "WildClawBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500", "title": "WildClawBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-winogrande", "domain": "reasoning", "name": "Winogrande", "released": "2019-11-21", "url": "https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500", "title": "Winogrande"}, {"benchmarks": [{"benchmark_id": "llm-stats-wmdp", "domain": "safety", "name": "WMDP", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500", "title": "WMDP"}, {"benchmarks": [{"benchmark_id": "llm-stats-wmt23", "domain": "language", "name": "WMT23", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500", "title": "WMT23"}, {"benchmarks": [{"benchmark_id": "llm-stats-wmt24", "domain": "language", "name": "WMT24++", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500", "title": "WMT24++"}, {"benchmarks": [{"benchmark_id": "llm-stats-workspace-bench", "domain": "reasoning", "name": "Workspace Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500", "title": "Workspace Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-worldbench", "domain": "multimodal", "name": "WorldBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500", "title": "WorldBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-worldvqa", "domain": "multimodal", "name": "WorldVQA", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500", "title": "WorldVQA"}, {"benchmarks": [{"benchmark_id": "llm-stats-writingbench", "domain": "legal", "name": "WritingBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500", "title": "WritingBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-xdailybench", "domain": "reasoning", "name": "xDailyBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500", "title": "xDailyBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-xlsum-english", "domain": "summarization", "name": "XLSum English", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500", "title": "XLSum English"}, {"benchmarks": [{"benchmark_id": "llm-stats-xstest", "domain": "safety", "name": "XSTest", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500", "title": "XSTest"}, {"benchmarks": [{"benchmark_id": "llm-stats-yc-bench", "domain": "finance", "name": "YC-Bench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500", "title": "YC-Bench"}, {"benchmarks": [{"benchmark_id": "llm-stats-zclawbench", "domain": "agents", "name": "ZClawBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500", "title": "ZClawBench"}, {"benchmarks": [{"benchmark_id": "llm-stats-zebralogic", "domain": "reasoning", "name": "ZebraLogic", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500", "title": "ZebraLogic"}, {"benchmarks": [{"benchmark_id": "llm-stats-zerobench-sub", "domain": "multimodal", "name": "ZEROBench-Sub", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500", "title": "ZEROBench-Sub"}, {"benchmarks": [{"benchmark_id": "llm-stats-zerobench", "domain": "multimodal", "name": "ZEROBench", "released": null, "url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500"}], "document_type": "registry_page", "id": "llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500", "source": "llm_stats", "source_url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500", "title": "ZEROBench"}, {"benchmarks": [{"benchmark_id": "agieval", "domain": "knowledge", "name": "AGIEval", "released": "2023-04-13", "url": "https://github.com/ruixiangcui/AGIEval"}, {"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "bfcl", "domain": "tool_use", "name": "BFCL", "released": "2024-02-26", "url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}, {"benchmark_id": "drop", "domain": "reasoning", "name": "DROP", "released": "2019-03-01", "url": "https://allenai.org/data/drop"}, {"benchmark_id": "gpqa", "domain": "science", "name": "GPQA (full)", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "gsm8k", "domain": "math", "name": "GSM8K", "released": "2021-10-27", "url": "https://github.com/openai/grade-school-math"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "ifbench", "domain": "instruction_following", "name": "IFBench", "released": "2025-07-03", "url": "https://arxiv.org/abs/2507.02833"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mbpp", "domain": "coding", "name": "MBPP", "released": "2021-08-16", "url": "https://github.com/google-research/google-research/tree/master/mbpp"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}], "document_type": "release_post", "id": "model_reports:ai2_olmo_3", "model_name": "Olmo 3 (7B and 32B)", "organization": "Ai2", "published": "2025-11-20", "retrieved_at": "2026-08-15", "source": "model_reports", "source_id": "ai2_olmo_3", "source_url": "https://allenai.org/blog/olmo3", "title": "Olmo 3 (7B and 32B)"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "tau_bench", "domain": "tool_use", "name": "tau-bench", "released": "2024-06-17", "url": "https://github.com/sierra-research/tau-bench"}], "document_type": "system_card", "id": "model_reports:anthropic_claude_3_7_sonnet", "model_name": "Claude 3.7 Sonnet", "organization": "Anthropic", "published": "2025-02-24", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "anthropic_claude_3_7_sonnet", "source_url": "https://www.anthropic.com/news/claude-3-7-sonnet", "title": "Claude 3.7 Sonnet"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "mmmlu", "domain": "multilingual", "name": "MMMLU", "released": "2024-09-24", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "tau_bench", "domain": "tool_use", "name": "tau-bench", "released": "2024-06-17", "url": "https://github.com/sierra-research/tau-bench"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "system_card", "id": "model_reports:anthropic_claude_4_system_card", "model_name": "Claude Opus 4 and Sonnet 4", "organization": "Anthropic", "published": "2025-05-22", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "anthropic_claude_4_system_card", "source_url": "https://www.anthropic.com/news/claude-4", "title": "Claude Opus 4 and Sonnet 4"}, {"benchmarks": [{"benchmark_id": "automationbench", "domain": "agent", "name": "AutomationBench", "released": "2026-03-10", "url": "https://github.com/zapier/automation-bench"}, {"benchmark_id": "biomysterybench", "domain": "biology", "name": "BioMysteryBench", "released": "2026-04-29", "url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-full"}, {"benchmark_id": "blueprint_bench_2", "domain": "spatial_reasoning", "name": "Blueprint-Bench 2", "released": "2026-05-04", "url": "https://andonlabs.com/evals/blueprint-bench-2"}, {"benchmark_id": "cursor_bench", "domain": "coding_agent", "name": "CursorBench", "released": "2025-10-29", "url": "https://cursor.com/blog/cursor-bench"}, {"benchmark_id": "exploitbench", "domain": "security", "name": "ExploitBench", "released": "2026-05-13", "url": "https://exploitbench.ai/"}, {"benchmark_id": "frontiercode", "domain": "coding_agent", "name": "FrontierCode", "released": "2026-06-08", "url": "https://cognition.com/blog/frontier-code"}, {"benchmark_id": "gdp_pdf", "domain": "multimodal", "name": "GDP.pdf", "released": "2026-04-14", "url": "https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world"}, {"benchmark_id": "gdpval", "domain": "professional", "name": "GDPval", "released": "2025-09-25", "url": "https://openai.com/index/gdpval/"}, {"benchmark_id": "healthbench_professional", "domain": "health", "name": "HealthBench Professional", "released": "2026-04-22", "url": "https://cdn.openai.com/dd128428-0184-4e25-b155-3a7686c7d744/HealthBench-Professional.pdf"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "legal_agent_benchmark", "domain": "legal", "name": "Legal Agent Benchmark", "released": "2026-05-06", "url": "https://www.harvey.ai/blog/introducing-harveys-legal-agent-benchmark"}, {"benchmark_id": "osworld", "domain": "computer_use", "name": "OSWorld", "released": "2024-04-11", "url": "https://os-world.github.io/"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "vibench", "domain": "coding_agent", "name": "ViBench", "released": "2026-05-26", "url": "https://vibench.ai/"}], "document_type": "release_post", "id": "model_reports:anthropic_claude_fable_5_mythos_5", "model_name": "Claude Fable 5 and Mythos 5", "organization": "Anthropic", "published": "2026-06-09", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "anthropic_claude_fable_5_mythos_5", "source_url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", "title": "Claude Fable 5 and Mythos 5"}, {"benchmarks": [{"benchmark_id": "arc_agi_3", "domain": "reasoning", "name": "ARC-AGI-3", "released": "2026-01-15", "url": "https://arcprize.org/arc-agi/3/"}, {"benchmark_id": "automationbench", "domain": "agent", "name": "AutomationBench", "released": "2026-03-10", "url": "https://github.com/zapier/automation-bench"}, {"benchmark_id": "cursor_bench", "domain": "coding_agent", "name": "CursorBench", "released": "2025-10-29", "url": "https://cursor.com/blog/cursor-bench"}, {"benchmark_id": "deepsearchqa", "domain": "agent", "name": "DeepSearchQA", "released": "2026-02-10", "url": "https://huggingface.co/datasets/PokeeAI/DeepSearchQA"}, {"benchmark_id": "gdpval", "domain": "professional", "name": "GDPval", "released": "2025-09-25", "url": "https://openai.com/index/gdpval/"}, {"benchmark_id": "osworld", "domain": "computer_use", "name": "OSWorld", "released": "2024-04-11", "url": "https://os-world.github.io/"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "system_card", "id": "model_reports:anthropic_claude_opus_5_system_card", "model_name": "Claude Opus 5", "organization": "Anthropic", "published": "2026-07-24", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "anthropic_claude_opus_5_system_card", "source_url": "https://www.anthropic.com/news/claude-opus-5", "title": "Claude Opus 5"}, {"benchmarks": [{"benchmark_id": "aider_polyglot", "domain": "coding", "name": "Aider Polyglot", "released": "2024-12-21", "url": "https://aider.chat/docs/leaderboards/"}, {"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "arena_hard", "domain": "human_preference", "name": "Arena-Hard", "released": "2024-04-19", "url": "https://github.com/lmarena/arena-hard-auto"}, {"benchmark_id": "drop", "domain": "reasoning", "name": "DROP", "released": "2019-03-01", "url": "https://allenai.org/data/drop"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmlu_redux", "domain": "knowledge", "name": "MMLU-Redux", "released": "2024-06-06", "url": "https://github.com/aryopg/mmlu-redux"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}], "document_type": "technical_report", "id": "model_reports:deepseek_r1_report", "model_name": "DeepSeek-R1", "organization": "DeepSeek", "published": "2025-01-22", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "deepseek_r1_report", "source_url": "https://arxiv.org/abs/2501.12948", "title": "DeepSeek-R1"}, {"benchmarks": [{"benchmark_id": "agieval", "domain": "knowledge", "name": "AGIEval", "released": "2023-04-13", "url": "https://github.com/ruixiangcui/AGIEval"}, {"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "arena_hard", "domain": "human_preference", "name": "Arena-Hard", "released": "2024-04-19", "url": "https://github.com/lmarena/arena-hard-auto"}, {"benchmark_id": "drop", "domain": "reasoning", "name": "DROP", "released": "2019-03-01", "url": "https://allenai.org/data/drop"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "gsm8k", "domain": "math", "name": "GSM8K", "released": "2021-10-27", "url": "https://github.com/openai/grade-school-math"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mbpp", "domain": "coding", "name": "MBPP", "released": "2021-08-16", "url": "https://github.com/google-research/google-research/tree/master/mbpp"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmlu_redux", "domain": "knowledge", "name": "MMLU-Redux", "released": "2024-06-06", "url": "https://github.com/aryopg/mmlu-redux"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}], "document_type": "technical_report", "id": "model_reports:deepseek_v3_report", "model_name": "DeepSeek-V3", "organization": "DeepSeek", "published": "2024-12-27", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "deepseek_v3_report", "source_url": "https://arxiv.org/abs/2412.19437", "title": "DeepSeek-V3"}, {"benchmarks": [{"benchmark_id": "agieval", "domain": "knowledge", "name": "AGIEval", "released": "2023-04-13", "url": "https://github.com/ruixiangcui/AGIEval"}, {"benchmark_id": "apex_agents", "domain": "agent", "name": "APEX-Agents", "released": "2026-01-20", "url": "https://arxiv.org/abs/2601.14242"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "codeforces", "domain": "coding", "name": "Codeforces", "released": "2010-02-19", "url": "https://codeforces.com/"}, {"benchmark_id": "drop", "domain": "reasoning", "name": "DROP", "released": "2019-03-01", "url": "https://allenai.org/data/drop"}, {"benchmark_id": "gdpval", "domain": "professional", "name": "GDPval", "released": "2025-09-25", "url": "https://openai.com/index/gdpval/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "gsm8k", "domain": "math", "name": "GSM8K", "released": "2021-10-27", "url": "https://github.com/openai/grade-school-math"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "hmmt", "domain": "math", "name": "HMMT", "released": "2025-02-15", "url": "https://www.hmmt.org/"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "imo_answer_bench", "domain": "math", "name": "IMOAnswerBench", "released": "2025-09-18", "url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "longbench", "domain": "long_context", "name": "LongBench", "released": "2023-08-28", "url": "https://github.com/THUDM/LongBench"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmlu_redux", "domain": "knowledge", "name": "MMLU-Redux", "released": "2024-06-06", "url": "https://github.com/aryopg/mmlu-redux"}, {"benchmark_id": "mmmlu", "domain": "multilingual", "name": "MMMLU", "released": "2024-09-24", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}, {"benchmark_id": "super_gpqa", "domain": "knowledge", "name": "SuperGPQA", "released": "2025-02-20", "url": "https://arxiv.org/abs/2502.14739"}, {"benchmark_id": "swe_bench_multilingual", "domain": "coding_agent", "name": "SWE-bench Multilingual", "released": "2025-04-22", "url": "https://www.swebench.com/multilingual.html"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}], "document_type": "model_card", "id": "model_reports:deepseek_v4_model_card", "model_name": "DeepSeek-V4-Pro", "organization": "DeepSeek", "published": "2026-04-22", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "deepseek_v4_model_card", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro", "title": "DeepSeek-V4-Pro"}, {"benchmarks": [{"benchmark_id": "automationbench", "domain": "agent", "name": "AutomationBench", "released": "2026-03-10", "url": "https://github.com/zapier/automation-bench"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}], "document_type": "model_card", "id": "model_reports:deepseek_v4_pro_0813_model_card", "model_name": "DeepSeek-V4-Pro-0813", "organization": "DeepSeek", "published": "2026-08-13", "retrieved_at": "2026-08-15", "source": "model_reports", "source_id": "deepseek_v4_pro_0813_model_card", "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813", "title": "DeepSeek-V4-Pro-0813"}, {"benchmarks": [{"benchmark_id": "apex_agents", "domain": "agent", "name": "APEX-Agents", "released": "2026-01-20", "url": "https://arxiv.org/abs/2601.14242"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "codeforces", "domain": "coding", "name": "Codeforces", "released": "2010-02-19", "url": "https://codeforces.com/"}, {"benchmark_id": "gdpval", "domain": "professional", "name": "GDPval", "released": "2025-09-25", "url": "https://openai.com/index/gdpval/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "hmmt", "domain": "math", "name": "HMMT", "released": "2025-02-15", "url": "https://www.hmmt.org/"}, {"benchmark_id": "imo_answer_bench", "domain": "math", "name": "IMOAnswerBench", "released": "2025-09-18", "url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "longbench", "domain": "long_context", "name": "LongBench", "released": "2023-08-28", "url": "https://github.com/THUDM/LongBench"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}, {"benchmark_id": "swe_bench_multilingual", "domain": "coding_agent", "name": "SWE-bench Multilingual", "released": "2025-04-22", "url": "https://www.swebench.com/multilingual.html"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}], "document_type": "technical_report", "id": "model_reports:deepseek_v4_technical_report", "model_name": "DeepSeek-V4", "organization": "DeepSeek", "published": "2026-06-25", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "deepseek_v4_technical_report", "source_url": "https://arxiv.org/abs/2606.19348", "title": "DeepSeek-V4"}, {"benchmarks": [{"benchmark_id": "frontier_challenge", "domain": "scientific_agent", "name": "FrontierChallenge", "released": "2026-08-25", "url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/"}], "document_type": "benchmark_leaderboard", "id": "model_reports:frontier_challenge_leaderboard_2026_08_25", "model_name": null, "organization": "ApodexAI", "published": null, "retrieved_at": "2026-09-02", "source": "model_reports", "source_id": "frontier_challenge_leaderboard_2026_08_25", "source_url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/", "title": "FrontierChallenge leaderboard"}, {"benchmarks": [{"benchmark_id": "chartqa", "domain": "multimodal", "name": "ChartQA", "released": "2022-03-19", "url": "https://github.com/vis-nlp/ChartQA"}, {"benchmark_id": "docvqa", "domain": "multimodal", "name": "DocVQA", "released": "2020-07-01", "url": "https://www.docvqa.org/"}, {"benchmark_id": "drop", "domain": "reasoning", "name": "DROP", "released": "2019-03-01", "url": "https://allenai.org/data/drop"}, {"benchmark_id": "gsm8k", "domain": "math", "name": "GSM8K", "released": "2021-10-27", "url": "https://github.com/openai/grade-school-math"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mathvista", "domain": "multimodal", "name": "MathVista", "released": "2023-10-03", "url": "https://mathvista.github.io/"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}], "document_type": "technical_report", "id": "model_reports:google_gemini_1_5_report", "model_name": "Gemini 1.5", "organization": "Google", "published": "2024-03-08", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "google_gemini_1_5_report", "source_url": "https://arxiv.org/abs/2403.05530", "title": "Gemini 1.5"}, {"benchmarks": [{"benchmark_id": "aider_polyglot", "domain": "coding", "name": "Aider Polyglot", "released": "2024-12-21", "url": "https://aider.chat/docs/leaderboards/"}, {"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "arc_agi", "domain": "reasoning", "name": "ARC-AGI", "released": "2019-11-05", "url": "https://arcprize.org/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mathvista", "domain": "multimodal", "name": "MathVista", "released": "2023-10-03", "url": "https://mathvista.github.io/"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "video_mme", "domain": "multimodal", "name": "Video-MME", "released": "2024-05-31", "url": "https://video-mme.github.io/"}], "document_type": "technical_report", "id": "model_reports:google_gemini_2_5_report", "model_name": "Gemini 2.5 Pro", "organization": "Google", "published": "2025-06-17", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "google_gemini_2_5_report", "source_url": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf", "title": "Gemini 2.5 Pro"}, {"benchmarks": [{"benchmark_id": "apex_agents", "domain": "agent", "name": "APEX-Agents", "released": "2026-01-20", "url": "https://arxiv.org/abs/2601.14242"}, {"benchmark_id": "arc_agi_2", "domain": "reasoning", "name": "ARC-AGI-2", "released": "2025-03-24", "url": "https://arcprize.org/arc-agi/2/"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "livecodebench_pro", "domain": "coding", "name": "LiveCodeBench Pro", "released": "2025-06-13", "url": "https://livecodebenchpro.com/"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "mmmlu", "domain": "multilingual", "name": "MMMLU", "released": "2024-09-24", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"benchmark_id": "mmmu_pro", "domain": "multimodal", "name": "MMMU-Pro", "released": "2024-09-04", "url": "https://arxiv.org/abs/2409.02813"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "scicode", "domain": "science", "name": "SciCode", "released": "2024-07-18", "url": "https://scicode-bench.github.io/"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "tau2_bench", "domain": "tool_use", "name": "tau2-bench", "released": "2025-06-09", "url": "https://arxiv.org/abs/2506.07982"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "model_card", "id": "model_reports:google_gemini_3_1_pro_model_card", "model_name": "Gemini 3.1 Pro", "organization": "Google", "published": "2026-02-19", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "google_gemini_3_1_pro_model_card", "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/", "title": "Gemini 3.1 Pro"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "bigbench_extra_hard", "domain": "reasoning", "name": "BIG-Bench Extra Hard", "released": "2025-02-26", "url": "https://arxiv.org/abs/2502.19187"}, {"benchmark_id": "codeforces", "domain": "coding", "name": "Codeforces", "released": "2010-02-19", "url": "https://codeforces.com/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mathvision", "domain": "multimodal", "name": "MathVision", "released": "2024-02-22", "url": "https://mathllm.github.io/mathvision/"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmmlu", "domain": "multilingual", "name": "MMMLU", "released": "2024-09-24", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"benchmark_id": "mmmu_pro", "domain": "multimodal", "name": "MMMU-Pro", "released": "2024-09-04", "url": "https://arxiv.org/abs/2409.02813"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "omnidocbench", "domain": "multimodal", "name": "OmniDocBench", "released": "2024-12-10", "url": "https://github.com/opendatalab/OmniDocBench"}, {"benchmark_id": "tau2_bench", "domain": "tool_use", "name": "tau2-bench", "released": "2025-06-09", "url": "https://arxiv.org/abs/2506.07982"}], "document_type": "model_card", "id": "model_reports:google_gemma_4_model_card", "model_name": "Gemma 4 (31B)", "organization": "Google", "published": "2026-03-11", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "google_gemma_4_model_card", "source_url": "https://huggingface.co/google/gemma-4-31B-it", "title": "Gemma 4 (31B)"}, {"benchmarks": [{"benchmark_id": "agieval", "domain": "knowledge", "name": "AGIEval", "released": "2023-04-13", "url": "https://github.com/ruixiangcui/AGIEval"}, {"benchmark_id": "arena_hard", "domain": "human_preference", "name": "Arena-Hard", "released": "2024-04-19", "url": "https://github.com/lmarena/arena-hard-auto"}, {"benchmark_id": "bfcl", "domain": "tool_use", "name": "BFCL", "released": "2024-02-26", "url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}, {"benchmark_id": "drop", "domain": "reasoning", "name": "DROP", "released": "2019-03-01", "url": "https://allenai.org/data/drop"}, {"benchmark_id": "gpqa", "domain": "science", "name": "GPQA (full)", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "gsm8k", "domain": "math", "name": "GSM8K", "released": "2021-10-27", "url": "https://github.com/openai/grade-school-math"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mbpp", "domain": "coding", "name": "MBPP", "released": "2021-08-16", "url": "https://github.com/google-research/google-research/tree/master/mbpp"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}], "document_type": "model_card", "id": "model_reports:meta_llama_3_1", "model_name": "Llama 3.1 405B", "organization": "Meta", "published": "2024-07-23", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "meta_llama_3_1", "source_url": "https://ai.meta.com/blog/meta-llama-3-1/", "title": "Llama 3.1 405B"}, {"benchmarks": [{"benchmark_id": "chartqa", "domain": "multimodal", "name": "ChartQA", "released": "2022-03-19", "url": "https://github.com/vis-nlp/ChartQA"}, {"benchmark_id": "chatbot_arena", "domain": "human_preference", "name": "Chatbot Arena", "released": "2023-05-03", "url": "https://lmarena.ai/"}, {"benchmark_id": "docvqa", "domain": "multimodal", "name": "DocVQA", "released": "2020-07-01", "url": "https://www.docvqa.org/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mathvista", "domain": "multimodal", "name": "MathVista", "released": "2023-10-03", "url": "https://mathvista.github.io/"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmmlu", "domain": "multilingual", "name": "MMMLU", "released": "2024-09-24", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "mtob", "domain": "long_context", "name": "MTOB", "released": "2023-09-28", "url": "https://arxiv.org/abs/2309.16575"}], "document_type": "model_card", "id": "model_reports:meta_llama_4", "model_name": "Llama 4", "organization": "Meta", "published": "2025-04-05", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "meta_llama_4", "source_url": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/", "title": "Llama 4"}, {"benchmarks": [{"benchmark_id": "gsm8k", "domain": "math", "name": "GSM8K", "released": "2021-10-27", "url": "https://github.com/openai/grade-school-math"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mbpp", "domain": "coding", "name": "MBPP", "released": "2021-08-16", "url": "https://github.com/google-research/google-research/tree/master/mbpp"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}], "document_type": "release_post", "id": "model_reports:mistral_large_2", "model_name": "Mistral Large 2", "organization": "Mistral", "published": "2024-07-24", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "mistral_large_2", "source_url": "https://mistral.ai/news/mistral-large-2407/", "title": "Mistral Large 2"}, {"benchmarks": [{"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mmmlu", "domain": "multilingual", "name": "MMMLU", "released": "2024-09-24", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}], "document_type": "model_card", "id": "model_reports:mistral_large_3", "model_name": "Mistral Large 3 (675B)", "organization": "Mistral", "published": "2025-11-28", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "mistral_large_3", "source_url": "https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512", "title": "Mistral Large 3 (675B)"}, {"benchmarks": [{"benchmark_id": "aider_polyglot", "domain": "coding", "name": "Aider Polyglot", "released": "2024-12-21", "url": "https://aider.chat/docs/leaderboards/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}], "document_type": "release_post", "id": "model_reports:mistral_medium_3", "model_name": "Mistral Medium 3", "organization": "Mistral", "published": "2025-05-07", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "mistral_medium_3", "source_url": "https://mistral.ai/news/mistral-medium-3", "title": "Mistral Medium 3"}, {"benchmarks": [{"benchmark_id": "aa_lcr", "domain": "long_context", "name": "AA-LCR", "released": "2025-09-16", "url": "https://artificialanalysis.ai/evaluations/aa-lcr"}, {"benchmark_id": "apex_agents", "domain": "agent", "name": "APEX-Agents", "released": "2026-01-20", "url": "https://arxiv.org/abs/2601.14242"}, {"benchmark_id": "automationbench", "domain": "agent", "name": "AutomationBench", "released": "2026-03-10", "url": "https://github.com/zapier/automation-bench"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "critpt", "domain": "science", "name": "CritPt", "released": "2025-09-30", "url": "https://arxiv.org/abs/2509.26574"}, {"benchmark_id": "deepsearchqa", "domain": "agent", "name": "DeepSearchQA", "released": "2026-02-10", "url": "https://huggingface.co/datasets/PokeeAI/DeepSearchQA"}, {"benchmark_id": "gdpval", "domain": "professional", "name": "GDPval", "released": "2025-09-25", "url": "https://openai.com/index/gdpval/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "mathvision", "domain": "multimodal", "name": "MathVision", "released": "2024-02-22", "url": "https://mathllm.github.io/mathvision/"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "mcp_mark", "domain": "tool_use", "name": "MCPMark", "released": "2025-09-30", "url": "https://mcpmark.ai/"}, {"benchmark_id": "mmmu_pro", "domain": "multimodal", "name": "MMMU-Pro", "released": "2024-09-04", "url": "https://arxiv.org/abs/2409.02813"}, {"benchmark_id": "omnidocbench", "domain": "multimodal", "name": "OmniDocBench", "released": "2024-12-10", "url": "https://github.com/opendatalab/OmniDocBench"}, {"benchmark_id": "osworld", "domain": "computer_use", "name": "OSWorld", "released": "2024-04-11", "url": "https://os-world.github.io/"}, {"benchmark_id": "posttrainbench", "domain": "ai_research", "name": "PostTrainBench", "released": "2026-01-20", "url": "https://posttrainbench.com/"}, {"benchmark_id": "scicode", "domain": "science", "name": "SciCode", "released": "2024-07-18", "url": "https://scicode-bench.github.io/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}, {"benchmark_id": "video_mme", "domain": "multimodal", "name": "Video-MME", "released": "2024-05-31", "url": "https://video-mme.github.io/"}], "document_type": "model_card", "id": "model_reports:moonshot_kimi_k3_model_card", "model_name": "Kimi K3", "organization": "Moonshot AI", "published": "2026-06-13", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "moonshot_kimi_k3_model_card", "source_url": "https://huggingface.co/moonshotai/Kimi-K3", "title": "Kimi K3"}, {"benchmarks": [{"benchmark_id": "aider_polyglot", "domain": "coding", "name": "Aider Polyglot", "released": "2024-12-21", "url": "https://aider.chat/docs/leaderboards/"}, {"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "video_mme", "domain": "multimodal", "name": "Video-MME", "released": "2024-05-31", "url": "https://video-mme.github.io/"}], "document_type": "release_post", "id": "model_reports:openai_gpt_4_1", "model_name": "GPT-4.1", "organization": "OpenAI", "published": "2025-04-14", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "openai_gpt_4_1", "source_url": "https://openai.com/index/gpt-4-1/", "title": "GPT-4.1"}, {"benchmarks": [{"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "release_post", "id": "model_reports:openai_gpt_5_6_release", "model_name": "GPT-5.6", "organization": "OpenAI", "published": "2026-06-26", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "openai_gpt_5_6_release", "source_url": "https://openai.com/index/gpt-5-6/", "title": "GPT-5.6"}, {"benchmarks": [{"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "cvebench", "domain": "security", "name": "CVE-Bench", "released": "2025-03-21", "url": "https://github.com/uiuc-kang-lab/cve-bench"}, {"benchmark_id": "exploitbench", "domain": "security", "name": "ExploitBench", "released": "2026-05-13", "url": "https://exploitbench.ai/"}, {"benchmark_id": "healthbench", "domain": "health", "name": "HealthBench", "released": "2025-05-12", "url": "https://openai.com/index/healthbench/"}, {"benchmark_id": "mle_bench", "domain": "ai_research", "name": "MLE-bench", "released": "2024-10-09", "url": "https://openai.com/index/mle-bench/"}, {"benchmark_id": "posttrainbench", "domain": "ai_research", "name": "PostTrainBench", "released": "2026-01-20", "url": "https://posttrainbench.com/"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "system_card", "id": "model_reports:openai_gpt_5_6_system_card", "model_name": "GPT-5.6 (Sol, Terra, Luna)", "organization": "OpenAI", "published": "2026-07-09", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "openai_gpt_5_6_system_card", "source_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "title": "GPT-5.6 (Sol, Terra, Luna)"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "healthbench", "domain": "health", "name": "HealthBench", "released": "2025-05-12", "url": "https://openai.com/index/healthbench/"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mle_bench", "domain": "ai_research", "name": "MLE-bench", "released": "2024-10-09", "url": "https://openai.com/index/mle-bench/"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "mrcr", "domain": "long_context", "name": "MRCR", "released": "2025-04-14", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "tau_bench", "domain": "tool_use", "name": "tau-bench", "released": "2024-06-17", "url": "https://github.com/sierra-research/tau-bench"}], "document_type": "system_card", "id": "model_reports:openai_gpt_5_system_card", "model_name": "GPT-5", "organization": "OpenAI", "published": "2025-08-07", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "openai_gpt_5_system_card", "source_url": "https://openai.com/index/gpt-5-system-card/", "title": "GPT-5"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "arc_agi", "domain": "reasoning", "name": "ARC-AGI", "released": "2019-11-05", "url": "https://arcprize.org/"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "simpleqa", "domain": "factuality", "name": "SimpleQA", "released": "2024-10-30", "url": "https://openai.com/index/introducing-simpleqa/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}], "document_type": "system_card", "id": "model_reports:openai_o3_o4_mini_system_card", "model_name": "o3 and o4-mini", "organization": "OpenAI", "published": "2025-04-16", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "openai_o3_o4_mini_system_card", "source_url": "https://openai.com/index/o3-o4-mini-system-card/", "title": "o3 and o4-mini"}, {"benchmarks": [{"benchmark_id": "gsm8k", "domain": "math", "name": "GSM8K", "released": "2021-10-27", "url": "https://github.com/openai/grade-school-math"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mbpp", "domain": "coding", "name": "MBPP", "released": "2021-08-16", "url": "https://github.com/google-research/google-research/tree/master/mbpp"}, {"benchmark_id": "mmlu", "domain": "knowledge", "name": "MMLU", "released": "2020-09-07", "url": "https://github.com/hendrycks/test"}], "document_type": "technical_report", "id": "model_reports:qwen2_5_coder_report", "model_name": "Qwen2.5-Coder", "organization": "Qwen", "published": "2024-09-18", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "qwen2_5_coder_report", "source_url": "https://arxiv.org/abs/2409.12186", "title": "Qwen2.5-Coder"}, {"benchmarks": [{"benchmark_id": "aa_lcr", "domain": "long_context", "name": "AA-LCR", "released": "2025-09-16", "url": "https://artificialanalysis.ai/evaluations/aa-lcr"}, {"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "bfcl", "domain": "tool_use", "name": "BFCL", "released": "2024-02-26", "url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "browsecomp_zh", "domain": "agent", "name": "BrowseComp-ZH", "released": "2025-04-27", "url": "https://arxiv.org/abs/2504.19314"}, {"benchmark_id": "gpqa", "domain": "science", "name": "GPQA (full)", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "hmmt", "domain": "math", "name": "HMMT", "released": "2025-02-15", "url": "https://www.hmmt.org/"}, {"benchmark_id": "ifbench", "domain": "instruction_following", "name": "IFBench", "released": "2025-07-03", "url": "https://arxiv.org/abs/2507.02833"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "imo_answer_bench", "domain": "math", "name": "IMOAnswerBench", "released": "2025-09-18", "url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "longbench", "domain": "long_context", "name": "LongBench", "released": "2023-08-28", "url": "https://github.com/THUDM/LongBench"}, {"benchmark_id": "mathvision", "domain": "multimodal", "name": "MathVision", "released": "2024-02-22", "url": "https://mathllm.github.io/mathvision/"}, {"benchmark_id": "mathvista", "domain": "multimodal", "name": "MathVista", "released": "2023-10-03", "url": "https://mathvista.github.io/"}, {"benchmark_id": "mcp_mark", "domain": "tool_use", "name": "MCPMark", "released": "2025-09-30", "url": "https://mcpmark.ai/"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmlu_redux", "domain": "knowledge", "name": "MMLU-Redux", "released": "2024-06-06", "url": "https://github.com/aryopg/mmlu-redux"}, {"benchmark_id": "mmmlu", "domain": "multilingual", "name": "MMMLU", "released": "2024-09-24", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "mmmu_pro", "domain": "multimodal", "name": "MMMU-Pro", "released": "2024-09-04", "url": "https://arxiv.org/abs/2409.02813"}, {"benchmark_id": "multichallenge", "domain": "instruction_following", "name": "MultiChallenge", "released": "2025-01-29", "url": "https://arxiv.org/abs/2501.17399"}, {"benchmark_id": "omnidocbench", "domain": "multimodal", "name": "OmniDocBench", "released": "2024-12-10", "url": "https://github.com/opendatalab/OmniDocBench"}, {"benchmark_id": "osworld", "domain": "computer_use", "name": "OSWorld", "released": "2024-04-11", "url": "https://os-world.github.io/"}, {"benchmark_id": "seccodebench", "domain": "security", "name": "SecCodeBench", "released": "2025-10-14", "url": "https://github.com/alibaba/SecCodeBench"}, {"benchmark_id": "super_gpqa", "domain": "knowledge", "name": "SuperGPQA", "released": "2025-02-20", "url": "https://arxiv.org/abs/2502.14739"}, {"benchmark_id": "swe_bench_multilingual", "domain": "coding_agent", "name": "SWE-bench Multilingual", "released": "2025-04-22", "url": "https://www.swebench.com/multilingual.html"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "tau2_bench", "domain": "tool_use", "name": "tau2-bench", "released": "2025-06-09", "url": "https://arxiv.org/abs/2506.07982"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}, {"benchmark_id": "video_mme", "domain": "multimodal", "name": "Video-MME", "released": "2024-05-31", "url": "https://video-mme.github.io/"}, {"benchmark_id": "widesearch", "domain": "agent", "name": "WideSearch", "released": "2025-08-11", "url": "https://widesearch-seed.github.io/"}], "document_type": "model_card", "id": "model_reports:qwen3_5_model_card", "model_name": "Qwen3.5-397B-A17B", "organization": "Qwen", "published": "2026-02-16", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "qwen3_5_model_card", "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", "title": "Qwen3.5-397B-A17B"}, {"benchmarks": [{"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "ifbench", "domain": "instruction_following", "name": "IFBench", "released": "2025-07-03", "url": "https://arxiv.org/abs/2507.02833"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mathvision", "domain": "multimodal", "name": "MathVision", "released": "2024-02-22", "url": "https://mathllm.github.io/mathvision/"}, {"benchmark_id": "omnidocbench", "domain": "multimodal", "name": "OmniDocBench", "released": "2024-12-10", "url": "https://github.com/opendatalab/OmniDocBench"}, {"benchmark_id": "osworld", "domain": "computer_use", "name": "OSWorld", "released": "2024-04-11", "url": "https://os-world.github.io/"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "model_card", "id": "model_reports:qwen3_8_model_card", "model_name": "Qwen3.8-27B", "organization": "Qwen", "published": "2026-08-05", "retrieved_at": "2026-08-15", "source": "model_reports", "source_id": "qwen3_8_model_card", "source_url": "https://ollama.com/library/qwen3.8", "title": "Qwen3.8-27B"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "arena_hard", "domain": "human_preference", "name": "Arena-Hard", "released": "2024-04-19", "url": "https://github.com/lmarena/arena-hard-auto"}, {"benchmark_id": "bfcl", "domain": "tool_use", "name": "BFCL", "released": "2024-02-26", "url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "humaneval", "domain": "coding", "name": "HumanEval", "released": "2021-07-07", "url": "https://github.com/openai/human-eval"}, {"benchmark_id": "ifeval", "domain": "instruction_following", "name": "IFEval", "released": "2023-11-14", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"benchmark_id": "livebench", "domain": "general", "name": "LiveBench", "released": "2024-06-06", "url": "https://livebench.ai/"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "math_500", "domain": "math", "name": "MATH-500", "released": "2021-03-05", "url": "https://github.com/openai/prm800k"}, {"benchmark_id": "mmlu_pro", "domain": "knowledge", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"benchmark_id": "mmlu_redux", "domain": "knowledge", "name": "MMLU-Redux", "released": "2024-06-06", "url": "https://github.com/aryopg/mmlu-redux"}], "document_type": "technical_report", "id": "model_reports:qwen3_technical_report", "model_name": "Qwen3", "organization": "Qwen", "published": "2025-05-14", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "qwen3_technical_report", "source_url": "https://arxiv.org/abs/2505.09388", "title": "Qwen3"}, {"benchmarks": [{"benchmark_id": "agents_last_exam", "domain": "coding_agent", "name": "Agents' Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "apex_agents", "domain": "agent", "name": "APEX-Agents", "released": "2026-01-20", "url": "https://arxiv.org/abs/2601.14242"}, {"benchmark_id": "arxivmath", "domain": "math", "name": "ArXivMath", "released": null, "url": "https://matharena.ai/arxivmath"}, {"benchmark_id": "automationbench", "domain": "agent", "name": "AutomationBench", "released": "2026-03-10", "url": "https://github.com/zapier/automation-bench"}, {"benchmark_id": "bankertoolbench", "domain": "professional", "name": "BankerToolBench", "released": null, "url": "https://github.com/Handshake-AI-Research/bankertoolbench"}, {"benchmark_id": "biomysterybench", "domain": "biology", "name": "BioMysteryBench", "released": "2026-04-29", "url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-full"}, {"benchmark_id": "brokenarxiv", "domain": "math", "name": "BrokenArXiv", "released": null, "url": "https://matharena.ai/brokenarxiv"}, {"benchmark_id": "critpt", "domain": "science", "name": "CritPt", "released": "2025-09-30", "url": "https://arxiv.org/abs/2509.26574"}, {"benchmark_id": "cybergym", "domain": "security", "name": "CyberGym", "released": null, "url": "https://openreview.net/forum?id=2YvbLQEdYt"}, {"benchmark_id": "deepswe", "domain": "coding_agent", "name": "DeepSWE", "released": "2025-09-15", "url": "https://github.com/deepswe/deepswe"}, {"benchmark_id": "draco", "domain": "long_context", "name": "DRACO", "released": null, "url": "https://huggingface.co/datasets/perplexity-ai/draco"}, {"benchmark_id": "gdpval", "domain": "professional", "name": "GDPval", "released": "2025-09-25", "url": "https://openai.com/index/gdpval/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "harbor_index", "domain": "coding_agent", "name": "Harbor-Index", "released": null, "url": "https://github.com/harbor-framework/harbor-index"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "horizonmath", "domain": "math", "name": "HorizonMath", "released": null, "url": "https://github.com/ewang26/HorizonMath"}, {"benchmark_id": "jobbench", "domain": "professional", "name": "JobBench", "released": "2026-05-25", "url": "https://arxiv.org/abs/2605.26329"}, {"benchmark_id": "matharena_apex_2025", "domain": "math", "name": "MathArena Apex 2025", "released": null, "url": "https://matharena.ai/apex"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "nl2repo", "domain": "coding_agent", "name": "NL2Repo", "released": "2025-06-15", "url": "https://github.com/nl2repo/nl2repo"}, {"benchmark_id": "office_qa_pro", "domain": "vision", "name": "OfficeQA Pro", "released": "2025-06-01", "url": "https://github.com/OfficeQA/OfficeQA-Pro"}, {"benchmark_id": "onemillionbench", "domain": "long_context", "name": "OneMillionBench", "released": null, "url": "https://github.com/humanlaya/OneMillion-Bench"}, {"benchmark_id": "posttrainbench", "domain": "ai_research", "name": "PostTrainBench", "released": "2026-01-20", "url": "https://posttrainbench.com/"}, {"benchmark_id": "programbench", "domain": "coding", "name": "ProgramBench", "released": "2026-05-05", "url": "https://arxiv.org/abs/2605.03546"}, {"benchmark_id": "skillsbench", "domain": "agent", "name": "SkillsBench", "released": "2026-02-13", "url": "https://skillsbench.ai/"}, {"benchmark_id": "superchem", "domain": "science", "name": "SUPERChem", "released": null, "url": "https://github.com/catalystforyou/SUPERChem_eval"}, {"benchmark_id": "swe_atlas", "domain": "coding_agent", "name": "SWE Atlas", "released": "2026-05-08", "url": "https://github.com/scaleapi/SWE-Atlas"}, {"benchmark_id": "swe_bench_multilingual", "domain": "coding_agent", "name": "SWE-bench Multilingual", "released": "2025-04-22", "url": "https://www.swebench.com/multilingual.html"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "swe_marathon", "domain": "coding_agent", "name": "SWE-Marathon", "released": null, "url": "https://github.com/abundant-ai/swe-marathon"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "toolathlon_verified", "domain": "tool_use", "name": "Toolathlon Verified", "released": "2025-11-01", "url": "https://github.com/toolathlon/toolathlon"}, {"benchmark_id": "widesearch", "domain": "agent", "name": "WideSearch", "released": "2025-08-11", "url": "https://widesearch-seed.github.io/"}, {"benchmark_id": "workspacebench", "domain": "professional", "name": "WorkspaceBench", "released": "2026-05-05", "url": "https://arxiv.org/abs/2605.03596"}], "document_type": "technical_report", "id": "model_reports:tencent_hy4_preview", "model_name": "Hy4 preview", "organization": "Tencent", "published": "2026-08-28", "retrieved_at": "2026-08-31", "source": "model_reports", "source_id": "tencent_hy4_preview", "source_url": "https://hy.tencent.ai/research/hy4-preview", "title": "Hy4 preview"}, {"benchmarks": [{"benchmark_id": "arc_agi_2", "domain": "reasoning", "name": "ARC-AGI-2", "released": "2025-03-24", "url": "https://arcprize.org/arc-agi/2/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "release_post", "id": "model_reports:xai_grok_4_5", "model_name": "Grok 4.5", "organization": "xAI", "published": "2026-07-08", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "xai_grok_4_5", "source_url": "https://x.ai/news/grok-4-5", "title": "Grok 4.5"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "arc_agi", "domain": "reasoning", "name": "ARC-AGI", "released": "2019-11-05", "url": "https://arcprize.org/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "livecodebench", "domain": "coding", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://livecodebench.github.io/"}, {"benchmark_id": "mmmu", "domain": "multimodal", "name": "MMMU", "released": "2023-11-27", "url": "https://mmmu-benchmark.github.io/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}], "document_type": "model_card", "id": "model_reports:xai_grok_4_model_card", "model_name": "Grok 4", "organization": "xAI", "published": "2025-07-09", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "xai_grok_4_model_card", "source_url": "https://x.ai/news/grok-4", "title": "Grok 4"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "hmmt", "domain": "math", "name": "HMMT", "released": "2025-02-15", "url": "https://www.hmmt.org/"}, {"benchmark_id": "imo_answer_bench", "domain": "math", "name": "IMOAnswerBench", "released": "2025-09-18", "url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "tau2_bench", "domain": "tool_use", "name": "tau2-bench", "released": "2025-06-09", "url": "https://arxiv.org/abs/2506.07982"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}, {"benchmark_id": "vending_bench", "domain": "agent", "name": "Vending Bench", "released": "2025-02-18", "url": "https://andonlabs.com/evals/vending-bench-2"}], "document_type": "model_card", "id": "model_reports:zai_glm_5_1_model_card", "model_name": "GLM-5.1", "organization": "Z.ai", "published": "2026-04-03", "retrieved_at": "2026-08-15", "source": "model_reports", "source_id": "zai_glm_5_1_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5.1", "title": "GLM-5.1"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "critpt", "domain": "science", "name": "CritPt", "released": "2025-09-30", "url": "https://arxiv.org/abs/2509.26574"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "hmmt", "domain": "math", "name": "HMMT", "released": "2025-02-15", "url": "https://www.hmmt.org/"}, {"benchmark_id": "imo_answer_bench", "domain": "math", "name": "IMOAnswerBench", "released": "2025-09-18", "url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "posttrainbench", "domain": "ai_research", "name": "PostTrainBench", "released": "2026-01-20", "url": "https://posttrainbench.com/"}, {"benchmark_id": "swe_bench_pro", "domain": "coding_agent", "name": "SWE-bench Pro", "released": "2025-09-23", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}], "document_type": "model_card", "id": "model_reports:zai_glm_5_2_model_card", "model_name": "GLM-5.2", "organization": "Z.ai", "published": "2026-06-16", "retrieved_at": "2026-08-15", "source": "model_reports", "source_id": "zai_glm_5_2_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5.2", "title": "GLM-5.2"}, {"benchmarks": [{"benchmark_id": "agents_last_exam", "domain": "coding_agent", "name": "Agents' Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "automationbench", "domain": "agent", "name": "AutomationBench", "released": "2026-03-10", "url": "https://github.com/zapier/automation-bench"}, {"benchmark_id": "babyvision", "domain": "multimodal", "name": "BabyVision", "released": "2025-12-01", "url": "https://github.com/babyvision/babyvision"}, {"benchmark_id": "chartography", "domain": "vision", "name": "Chartography", "released": "2025-06-01", "url": "https://github.com/Chartography/Chartography"}, {"benchmark_id": "charxiv_reasoning", "domain": "vision", "name": "CharXiv Reasoning", "released": "2025-06-01", "url": "https://github.com/CharXiv/CharXiv"}, {"benchmark_id": "deepswe", "domain": "coding_agent", "name": "DeepSWE", "released": "2025-09-15", "url": "https://github.com/deepswe/deepswe"}, {"benchmark_id": "gdpval", "domain": "professional", "name": "GDPval", "released": "2025-09-25", "url": "https://openai.com/index/gdpval/"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "mmvu", "domain": "vision", "name": "MMVU", "released": "2025-06-01", "url": "https://github.com/MMVU/MMVU"}, {"benchmark_id": "mvbench", "domain": "vision", "name": "MVbench", "released": "2025-06-01", "url": "https://github.com/MVBench/MVBench"}, {"benchmark_id": "nl2repo", "domain": "coding_agent", "name": "NL2Repo", "released": "2025-06-15", "url": "https://github.com/nl2repo/nl2repo"}, {"benchmark_id": "office_qa_pro", "domain": "vision", "name": "OfficeQA Pro", "released": "2025-06-01", "url": "https://github.com/OfficeQA/OfficeQA-Pro"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "toolathlon_verified", "domain": "tool_use", "name": "Toolathlon Verified", "released": "2025-11-01", "url": "https://github.com/toolathlon/toolathlon"}], "document_type": "model_card", "id": "model_reports:zai_glm_5_3_flash_model_card", "model_name": "GLM-5.3-Flash", "organization": "Z.ai", "published": "2026-08-27", "retrieved_at": "2026-08-27", "source": "model_reports", "source_id": "zai_glm_5_3_flash_model_card", "source_url": "https://z.ai/blog/glm-5.3-flash", "title": "GLM-5.3-Flash"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "browsecomp_zh", "domain": "agent", "name": "BrowseComp-ZH", "released": "2025-04-27", "url": "https://arxiv.org/abs/2504.19314"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "hmmt", "domain": "math", "name": "HMMT", "released": "2025-02-15", "url": "https://www.hmmt.org/"}, {"benchmark_id": "imo_answer_bench", "domain": "math", "name": "IMOAnswerBench", "released": "2025-09-18", "url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, {"benchmark_id": "mcp_atlas", "domain": "tool_use", "name": "MCP Atlas", "released": "2025-11-18", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"benchmark_id": "swe_bench_multilingual", "domain": "coding_agent", "name": "SWE-bench Multilingual", "released": "2025-04-22", "url": "https://www.swebench.com/multilingual.html"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "tau2_bench", "domain": "tool_use", "name": "tau2-bench", "released": "2025-06-09", "url": "https://arxiv.org/abs/2506.07982"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}, {"benchmark_id": "tool_decathlon", "domain": "tool_use", "name": "Tool Decathlon", "released": "2025-10-28", "url": "https://toolathlon.xyz/"}, {"benchmark_id": "vending_bench", "domain": "agent", "name": "Vending Bench", "released": "2025-02-18", "url": "https://andonlabs.com/evals/vending-bench-2"}], "document_type": "model_card", "id": "model_reports:zai_glm_5_model_card", "model_name": "GLM-5", "organization": "Z.ai", "published": "2026-02-11", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "zai_glm_5_model_card", "source_url": "https://huggingface.co/zai-org/GLM-5", "title": "GLM-5"}, {"benchmarks": [{"benchmark_id": "aime", "domain": "math", "name": "AIME", "released": "2024-02-01", "url": "https://maa.org/maa-invitational-competitions/"}, {"benchmark_id": "browsecomp", "domain": "agent", "name": "BrowseComp", "released": "2025-04-10", "url": "https://openai.com/index/browsecomp/"}, {"benchmark_id": "gpqa_diamond", "domain": "science", "name": "GPQA Diamond", "released": "2023-11-20", "url": "https://arxiv.org/abs/2311.12022"}, {"benchmark_id": "hle", "domain": "reasoning", "name": "Humanity's Last Exam", "released": "2025-01-23", "url": "https://lastexam.ai/"}, {"benchmark_id": "swe_bench_verified", "domain": "coding_agent", "name": "SWE-bench Verified", "released": "2024-08-13", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"benchmark_id": "tau2_bench", "domain": "tool_use", "name": "tau2-bench", "released": "2025-06-09", "url": "https://arxiv.org/abs/2506.07982"}, {"benchmark_id": "terminal_bench", "domain": "agent", "name": "Terminal-Bench", "released": "2025-05-19", "url": "https://www.tbench.ai/"}], "document_type": "technical_report", "id": "model_reports:zai_glm_5_paper", "model_name": "GLM-5", "organization": "Z.ai", "published": "2026-02-11", "retrieved_at": "2026-08-02", "source": "model_reports", "source_id": "zai_glm_5_paper", "source_url": "https://huggingface.co/papers/2602.15763", "title": "GLM-5"}, {"benchmarks": [{"benchmark_id": "opencompass-1580-a-bench", "domain": "多模态", "name": "A-Bench", "released": "2024-06-05", "url": "https://hub.opencompass.org.cn/dataset-detail/A-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/A-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/A-Bench", "title": "A-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1367-a-okvqa", "domain": "多模态", "name": "A-OKVQA", "released": "2022-06-03", "url": "https://hub.opencompass.org.cn/dataset-detail/A-OKVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/A-OKVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/A-OKVQA", "title": "A-OKVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-2452-aecbench", "domain": "科学", "name": "AECBench", "released": "2026-04-10", "url": "https://hub.opencompass.org.cn/dataset-detail/AECBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AECBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AECBench", "title": "AECBench"}, {"benchmarks": [{"benchmark_id": "opencompass-506-afqmc", "domain": "语言", "name": "AFQMC", "released": "2020-04-13", "url": "https://hub.opencompass.org.cn/dataset-detail/AFQMC"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AFQMC", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AFQMC", "title": "AFQMC"}, {"benchmarks": [{"benchmark_id": "opencompass-497-agieval", "domain": "学科", "name": "AGIEval", "released": "2023-09-18", "url": "https://hub.opencompass.org.cn/dataset-detail/AGIEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AGIEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AGIEval", "title": "AGIEval"}, {"benchmarks": [{"benchmark_id": "opencompass-2396-aidabench", "domain": "多模态", "name": "AIDABench", "released": "2026-02-14", "url": "https://hub.opencompass.org.cn/dataset-detail/AIDABench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIDABench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIDABench", "title": "AIDABench"}, {"benchmarks": [{"benchmark_id": "opencompass-1069-air-bench", "domain": "知识", "name": "AIR-Bench", "released": "2024-02-12", "url": "https://hub.opencompass.org.cn/dataset-detail/AIR-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIR-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIR-Bench", "title": "AIR-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1557-airbench-2024", "domain": "安全", "name": "AIRBench-2024", "released": "2024-08-05", "url": "https://hub.opencompass.org.cn/dataset-detail/AIRBench-2024"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIRBench-2024", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIRBench-2024", "title": "AIRBench-2024"}, {"benchmarks": [{"benchmark_id": "opencompass-1995-airtbench", "domain": "智能体", "name": "AIRTBench", "released": "2025-06-17", "url": "https://hub.opencompass.org.cn/dataset-detail/AIRTBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIRTBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AIRTBench", "title": "AIRTBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1969-ale-bench", "domain": "代码", "name": "ALE-Bench", "released": "2025-06-10", "url": "https://hub.opencompass.org.cn/dataset-detail/ALE-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ALE-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ALE-Bench", "title": "ALE-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1280-ambrosia", "domain": "推理", "name": "AMBROSIA", "released": "2024-06-27", "url": "https://hub.opencompass.org.cn/dataset-detail/AMBROSIA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AMBROSIA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AMBROSIA", "title": "AMBROSIA"}, {"benchmarks": [{"benchmark_id": "opencompass-1966-amsbench", "domain": "多模态", "name": "AMSbench", "released": "2025-06-21", "url": "https://hub.opencompass.org.cn/dataset-detail/AMSbench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AMSbench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AMSbench", "title": "AMSbench"}, {"benchmarks": [{"benchmark_id": "opencompass-1093-apps", "domain": "代码", "name": "APPS", "released": "2021-11-08", "url": "https://hub.opencompass.org.cn/dataset-detail/APPS"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/APPS", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/APPS", "title": "APPS"}, {"benchmarks": [{"benchmark_id": "opencompass-1116-aqua-rat", "domain": "数学", "name": "AQUA-RAT", "released": "2017-10-23", "url": "https://hub.opencompass.org.cn/dataset-detail/AQUA-RAT"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AQUA-RAT", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AQUA-RAT", "title": "AQUA-RAT"}, {"benchmarks": [{"benchmark_id": "opencompass-502-arc-c", "domain": "学科", "name": "ARC-c", "released": "2018-03-14", "url": "https://hub.opencompass.org.cn/dataset-detail/ARC-c"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ARC-c", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ARC-c", "title": "ARC-c"}, {"benchmarks": [{"benchmark_id": "opencompass-503-arc-e", "domain": "学科", "name": "ARC-e", "released": "2018-03-14", "url": "https://hub.opencompass.org.cn/dataset-detail/ARC-e"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ARC-e", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ARC-e", "title": "ARC-e"}, {"benchmarks": [{"benchmark_id": "opencompass-1114-asdiv", "domain": "数学", "name": "ASDiv", "released": "2020-07-05", "url": "https://hub.opencompass.org.cn/dataset-detail/ASDiv"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ASDiv", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ASDiv", "title": "ASDiv"}, {"benchmarks": [{"benchmark_id": "opencompass-1396-av-odyssey-bench", "domain": "多模态", "name": "AV-Odyssey-Bench", "released": "2024-12-03", "url": "https://hub.opencompass.org.cn/dataset-detail/AV-Odyssey-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AV-Odyssey-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AV-Odyssey-Bench", "title": "AV-Odyssey-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-526-ax-b", "domain": "推理", "name": "AX-b", "released": "2019-05-02", "url": "https://hub.opencompass.org.cn/dataset-detail/AX-b"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AX-b", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AX-b", "title": "AX-b"}, {"benchmarks": [{"benchmark_id": "opencompass-527-ax-g", "domain": "推理", "name": "AX-g", "released": "2019-07-12", "url": "https://hub.opencompass.org.cn/dataset-detail/AX-g"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AX-g", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AX-g", "title": "AX-g"}, {"benchmarks": [{"benchmark_id": "opencompass-2084-axbench", "domain": "理解", "name": "AXBENCH", "released": "2025-01-28", "url": "https://hub.opencompass.org.cn/dataset-detail/AXBENCH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AXBENCH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AXBENCH", "title": "AXBENCH"}, {"benchmarks": [{"benchmark_id": "opencompass-1148-abspyramid", "domain": "知识", "name": "AbsPyramid", "released": "2024-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/AbsPyramid"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AbsPyramid", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AbsPyramid", "title": "AbsPyramid"}, {"benchmarks": [{"benchmark_id": "opencompass-1319-actionatlas", "domain": "多模态", "name": "ActionAtlas", "released": "2024-10-08", "url": "https://hub.opencompass.org.cn/dataset-detail/ActionAtlas"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ActionAtlas", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ActionAtlas", "title": "ActionAtlas"}, {"benchmarks": [{"benchmark_id": "opencompass-1155-ada-leval", "domain": "长文本", "name": "Ada-LEval", "released": "2024-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/Ada-LEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Ada-LEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Ada-LEval", "title": "Ada-LEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1753-agmmu", "domain": "多模态", "name": "AgMMU", "released": "2025-04-14", "url": "https://hub.opencompass.org.cn/dataset-detail/AgMMU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgMMU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgMMU", "title": "AgMMU"}, {"benchmarks": [{"benchmark_id": "opencompass-1242-agentboard", "domain": "智能体", "name": "AgentBoard", "released": "2024-06-24", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentBoard"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentBoard", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentBoard", "title": "AgentBoard"}, {"benchmarks": [{"benchmark_id": "opencompass-1351-agentharm", "domain": "安全", "name": "AgentHarm", "released": "2024-10-11", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentHarm"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentHarm", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentHarm", "title": "AgentHarm"}, {"benchmarks": [{"benchmark_id": "opencompass-2061-agenthazard", "domain": "多模态", "name": "AgentHazard", "released": "2025-07-16", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentHazard"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentHazard", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentHazard", "title": "AgentHazard"}, {"benchmarks": [{"benchmark_id": "opencompass-1778-agentrewardbench", "domain": "推理", "name": "AgentRewardBench", "released": "2025-04-11", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentRewardBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentRewardBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AgentRewardBench", "title": "AgentRewardBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1075-alignbench", "domain": "理解", "name": "AlignBench", "released": "2024-08-25", "url": "https://hub.opencompass.org.cn/dataset-detail/AlignBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AlignBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AlignBench", "title": "AlignBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1876-aneumo", "domain": "多模态", "name": "Aneumo", "released": "2025-05-19", "url": "https://hub.opencompass.org.cn/dataset-detail/Aneumo"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Aneumo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Aneumo", "title": "Aneumo"}, {"benchmarks": [{"benchmark_id": "opencompass-2075-arena-hard-auto", "domain": "安全", "name": "Arena-Hard-Auto", "released": "2024-04-19", "url": "https://hub.opencompass.org.cn/dataset-detail/Arena-Hard-Auto"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Arena-Hard-Auto", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Arena-Hard-Auto", "title": "Arena-Hard-Auto"}, {"benchmarks": [{"benchmark_id": "opencompass-2370-argusinspection", "domain": "多模态", "name": "ArgusInspection", "released": "2025-10-27", "url": "https://hub.opencompass.org.cn/dataset-detail/ArgusInspection"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ArgusInspection", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ArgusInspection", "title": "ArgusInspection"}, {"benchmarks": [{"benchmark_id": "opencompass-2052-artifactsbench", "domain": "多模态", "name": "ArtifactsBench", "released": "2025-07-01", "url": "https://hub.opencompass.org.cn/dataset-detail/ArtifactsBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ArtifactsBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ArtifactsBench", "title": "ArtifactsBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1991-assetopsbench", "domain": "智能体", "name": "AssetOpsBench", "released": "2025-06-04", "url": "https://hub.opencompass.org.cn/dataset-detail/AssetOpsBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AssetOpsBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AssetOpsBench", "title": "AssetOpsBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1835-audiojailbreak", "domain": "多模态", "name": "AudioJailbreak", "released": "2025-05-24", "url": "https://hub.opencompass.org.cn/dataset-detail/AudioJailbreak"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AudioJailbreak", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AudioJailbreak", "title": "AudioJailbreak"}, {"benchmarks": [{"benchmark_id": "opencompass-1908-audiotrust", "domain": "多模态", "name": "AudioTrust", "released": "2025-06-06", "url": "https://hub.opencompass.org.cn/dataset-detail/AudioTrust"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AudioTrust", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AudioTrust", "title": "AudioTrust"}, {"benchmarks": [{"benchmark_id": "opencompass-2078-autoadvexbench", "domain": "安全", "name": "AutoAdvExBench", "released": "2025-03-03", "url": "https://hub.opencompass.org.cn/dataset-detail/AutoAdvExBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AutoAdvExBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/AutoAdvExBench", "title": "AutoAdvExBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1266-babilong", "domain": "长文本", "name": "BABILong", "released": "2024-06-14", "url": "https://hub.opencompass.org.cn/dataset-detail/BABILong"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BABILong", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BABILong", "title": "BABILong"}, {"benchmarks": [{"benchmark_id": "opencompass-539-bbh", "domain": "推理", "name": "BBH", "released": "2022-10-17", "url": "https://hub.opencompass.org.cn/dataset-detail/BBH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BBH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BBH", "title": "BBH"}, {"benchmarks": [{"benchmark_id": "opencompass-1365-blink", "domain": "多模态", "name": "BLINK", "released": "2024-04-28", "url": "https://hub.opencompass.org.cn/dataset-detail/BLINK"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BLINK", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BLINK", "title": "BLINK"}, {"benchmarks": [{"benchmark_id": "opencompass-1571-bright", "domain": "强推理", "name": "BRIGHT", "released": "2024-10-24", "url": "https://hub.opencompass.org.cn/dataset-detail/BRIGHT"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BRIGHT", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BRIGHT", "title": "BRIGHT"}, {"benchmarks": [{"benchmark_id": "opencompass-1146-bust", "domain": "理解", "name": "BUST", "released": "2024-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/BUST"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BUST", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BUST", "title": "BUST"}, {"benchmarks": [{"benchmark_id": "opencompass-1086-belebele", "domain": "语言", "name": "Belebele", "released": "2024-07-25", "url": "https://hub.opencompass.org.cn/dataset-detail/Belebele"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Belebele", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Belebele", "title": "Belebele"}, {"benchmarks": [{"benchmark_id": "opencompass-1562-benchmax", "domain": "语言", "name": "BenchMAX", "released": "2025-02-11", "url": "https://hub.opencompass.org.cn/dataset-detail/BenchMAX"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BenchMAX", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BenchMAX", "title": "BenchMAX"}, {"benchmarks": [{"benchmark_id": "opencompass-1253-bigcodebench", "domain": "强推理", "name": "BigCodeBench", "released": "2024-06-22", "url": "https://hub.opencompass.org.cn/dataset-detail/BigCodeBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BigCodeBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BigCodeBench", "title": "BigCodeBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1680-bigobench", "domain": "代码", "name": "BigOBench", "released": "2025-03-19", "url": "https://hub.opencompass.org.cn/dataset-detail/BigOBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BigOBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BigOBench", "title": "BigOBench"}, {"benchmarks": [{"benchmark_id": "opencompass-510-boolq", "domain": "知识", "name": "BoolQ", "released": "2019-05-24", "url": "https://hub.opencompass.org.cn/dataset-detail/BoolQ"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BoolQ", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/BoolQ", "title": "BoolQ"}, {"benchmarks": [{"benchmark_id": "opencompass-1983-bytemorph", "domain": "创作", "name": "ByteMorph", "released": "2025-06-03", "url": "https://hub.opencompass.org.cn/dataset-detail/ByteMorph"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ByteMorph", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ByteMorph", "title": "ByteMorph"}, {"benchmarks": [{"benchmark_id": "opencompass-496-c-eval", "domain": "学科", "name": "C-Eval", "released": "2023-05-15", "url": "https://hub.opencompass.org.cn/dataset-detail/C-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/C-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/C-Eval", "title": "C-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-1780-c-faith", "domain": "语言", "name": "C-FAITH", "released": "2025-04-14", "url": "https://hub.opencompass.org.cn/dataset-detail/C-FAITH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/C-FAITH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/C-FAITH", "title": "C-FAITH"}, {"benchmarks": [{"benchmark_id": "opencompass-514-c3", "domain": "理解", "name": "C3", "released": "2019-04-21", "url": "https://hub.opencompass.org.cn/dataset-detail/C3"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/C3", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/C3", "title": "C3"}, {"benchmarks": [{"benchmark_id": "opencompass-1552-ceb", "domain": "其他", "name": "CEB", "released": "2024-07-03", "url": "https://hub.opencompass.org.cn/dataset-detail/CEB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CEB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CEB", "title": "CEB"}, {"benchmarks": [{"benchmark_id": "opencompass-1079-cflue", "domain": "知识", "name": "CFLUE", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/CFLUE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CFLUE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CFLUE", "title": "CFLUE"}, {"benchmarks": [{"benchmark_id": "opencompass-1907-cfinbench", "domain": "知识", "name": "CFinBench", "released": "2024-10-01", "url": "https://hub.opencompass.org.cn/dataset-detail/CFinBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CFinBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CFinBench", "title": "CFinBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1513-cg-bench", "domain": "多模态", "name": "CG-Bench", "released": "2024-12-16", "url": "https://hub.opencompass.org.cn/dataset-detail/CG-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CG-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CG-Bench", "title": "CG-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1141-charm", "domain": "推理", "name": "CHARM", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/CHARM"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CHARM", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CHARM", "title": "CHARM"}, {"benchmarks": [{"benchmark_id": "opencompass-1534-chase-code", "domain": "代码", "name": "CHASE-Code", "released": "2025-02-14", "url": "https://hub.opencompass.org.cn/dataset-detail/CHASE-Code"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CHASE-Code", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CHASE-Code", "title": "CHASE-Code"}, {"benchmarks": [{"benchmark_id": "opencompass-505-chid", "domain": "语言", "name": "CHID", "released": "2019-06-04", "url": "https://hub.opencompass.org.cn/dataset-detail/CHID"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CHID", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CHID", "title": "CHID"}, {"benchmarks": [{"benchmark_id": "opencompass-1834-clever", "domain": "代码", "name": "CLEVER", "released": "2025-05-20", "url": "https://hub.opencompass.org.cn/dataset-detail/CLEVER"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CLEVER", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CLEVER", "title": "CLEVER"}, {"benchmarks": [{"benchmark_id": "opencompass-1137-cmb", "domain": "其他", "name": "CMB", "released": "2024-04-04", "url": "https://hub.opencompass.org.cn/dataset-detail/CMB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CMB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CMB", "title": "CMB"}, {"benchmarks": [{"benchmark_id": "opencompass-499-cmmlu", "domain": "学科", "name": "CMMLU", "released": "2023-06-15", "url": "https://hub.opencompass.org.cn/dataset-detail/CMMLU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CMMLU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CMMLU", "title": "CMMLU"}, {"benchmarks": [{"benchmark_id": "opencompass-524-cmnli", "domain": "推理", "name": "CMNLI", "released": null, "url": "https://hub.opencompass.org.cn/dataset-detail/CMNLI"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CMNLI", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CMNLI", "title": "CMNLI"}, {"benchmarks": [{"benchmark_id": "opencompass-529-copa", "domain": "推理", "name": "COPA", "released": null, "url": "https://hub.opencompass.org.cn/dataset-detail/COPA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/COPA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/COPA", "title": "COPA"}, {"benchmarks": [{"benchmark_id": "opencompass-1622-coral", "domain": "其他", "name": "CORAL", "released": "2024-10-30", "url": "https://hub.opencompass.org.cn/dataset-detail/CORAL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CORAL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CORAL", "title": "CORAL"}, {"benchmarks": [{"benchmark_id": "opencompass-1972-cpret", "domain": "代码", "name": "CPRet", "released": "2025-05-19", "url": "https://hub.opencompass.org.cn/dataset-detail/CPRet"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CPRet", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CPRet", "title": "CPRet"}, {"benchmarks": [{"benchmark_id": "opencompass-1626-crag", "domain": "其他", "name": "CRAG", "released": "2024-06-07", "url": "https://hub.opencompass.org.cn/dataset-detail/CRAG"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CRAG", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CRAG", "title": "CRAG"}, {"benchmarks": [{"benchmark_id": "opencompass-2046-crew-wildfire", "domain": "智能体", "name": "CREW-WILDFIRE", "released": "2025-07-07", "url": "https://hub.opencompass.org.cn/dataset-detail/CREW-WILDFIRE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CREW-WILDFIRE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CREW-WILDFIRE", "title": "CREW-WILDFIRE"}, {"benchmarks": [{"benchmark_id": "opencompass-1500-crpe", "domain": "多模态", "name": "CRPE", "released": "2024-02-29", "url": "https://hub.opencompass.org.cn/dataset-detail/CRPE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CRPE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CRPE", "title": "CRPE"}, {"benchmarks": [{"benchmark_id": "opencompass-924-cs-bench", "domain": "推理", "name": "CS-Bench", "released": "2024-06-12", "url": "https://hub.opencompass.org.cn/dataset-detail/CS-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CS-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CS-Bench", "title": "CS-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1219-cs-eval", "domain": "学科", "name": "CS-Eval", "released": "2024-05-31", "url": "https://hub.opencompass.org.cn/dataset-detail/CS-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CS-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CS-Eval", "title": "CS-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-519-csl", "domain": "理解", "name": "CSL", "released": "2022-09-12", "url": "https://hub.opencompass.org.cn/dataset-detail/CSL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CSL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CSL", "title": "CSL"}, {"benchmarks": [{"benchmark_id": "opencompass-1833-csts", "domain": "其他", "name": "CSTS", "released": "2025-05-20", "url": "https://hub.opencompass.org.cn/dataset-detail/CSTS"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CSTS", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CSTS", "title": "CSTS"}, {"benchmarks": [{"benchmark_id": "opencompass-1268-ctibench", "domain": "安全", "name": "CTIBench", "released": "2024-06-11", "url": "https://hub.opencompass.org.cn/dataset-detail/CTIBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CTIBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CTIBench", "title": "CTIBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1985-cvdp", "domain": "智能体", "name": "CVDP", "released": "2025-06-17", "url": "https://hub.opencompass.org.cn/dataset-detail/CVDP"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CVDP", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CVDP", "title": "CVDP"}, {"benchmarks": [{"benchmark_id": "opencompass-1239-cvqa", "domain": "多模态", "name": "CVQA", "released": "2024-06-10", "url": "https://hub.opencompass.org.cn/dataset-detail/CVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CVQA", "title": "CVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1052-calm", "domain": "推理", "name": "CaLM", "released": "2024-05-01", "url": "https://hub.opencompass.org.cn/dataset-detail/CaLM"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CaLM", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CaLM", "title": "CaLM"}, {"benchmarks": [{"benchmark_id": "opencompass-1975-causalvqa", "domain": "多模态", "name": "CausalVQA", "released": "2025-06-11", "url": "https://hub.opencompass.org.cn/dataset-detail/CausalVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CausalVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CausalVQA", "title": "CausalVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-692-chembench", "domain": "知识", "name": "ChemBench", "released": "2024-02-15", "url": "https://hub.opencompass.org.cn/dataset-detail/ChemBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ChemBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ChemBench", "title": "ChemBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1278-chronomagic-bench", "domain": "创作", "name": "ChronoMagic-Bench", "released": "2024-06-26", "url": "https://hub.opencompass.org.cn/dataset-detail/ChronoMagic-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ChronoMagic-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ChronoMagic-Bench", "title": "ChronoMagic-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1849-cleanpatrick", "domain": "理解", "name": "CleanPatrick", "released": "2025-05-16", "url": "https://hub.opencompass.org.cn/dataset-detail/CleanPatrick"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CleanPatrick", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CleanPatrick", "title": "CleanPatrick"}, {"benchmarks": [{"benchmark_id": "opencompass-1954-climateviz", "domain": "多模态", "name": "ClimateViz", "released": "2025-06-11", "url": "https://hub.opencompass.org.cn/dataset-detail/ClimateViz"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ClimateViz", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ClimateViz", "title": "ClimateViz"}, {"benchmarks": [{"benchmark_id": "opencompass-1546-codecriticbench", "domain": "代码", "name": "CodeCriticBench", "released": "2025-02-23", "url": "https://hub.opencompass.org.cn/dataset-detail/CodeCriticBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CodeCriticBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CodeCriticBench", "title": "CodeCriticBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1624-codeelo", "domain": "代码", "name": "CodeElo", "released": "2025-01-03", "url": "https://hub.opencompass.org.cn/dataset-detail/CodeElo"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CodeElo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CodeElo", "title": "CodeElo"}, {"benchmarks": [{"benchmark_id": "opencompass-1582-codemmlu", "domain": "代码", "name": "CodeMMLU", "released": "2024-06-22", "url": "https://hub.opencompass.org.cn/dataset-detail/CodeMMLU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CodeMMLU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CodeMMLU", "title": "CodeMMLU"}, {"benchmarks": [{"benchmark_id": "opencompass-1250-collie", "domain": "强推理", "name": "Collie", "released": "2023-07-17", "url": "https://hub.opencompass.org.cn/dataset-detail/Collie"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Collie", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Collie", "title": "Collie"}, {"benchmarks": [{"benchmark_id": "opencompass-1777-colorbench", "domain": "多模态", "name": "ColorBench", "released": "2025-04-10", "url": "https://hub.opencompass.org.cn/dataset-detail/ColorBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ColorBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ColorBench", "title": "ColorBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1943-combibench", "domain": "数学", "name": "CombiBench", "released": "2025-04-28", "url": "https://hub.opencompass.org.cn/dataset-detail/CombiBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CombiBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CombiBench", "title": "CombiBench"}, {"benchmarks": [{"benchmark_id": "opencompass-511-commonsenseqa", "domain": "知识", "name": "CommonSenseQA", "released": "2018-11-02", "url": "https://hub.opencompass.org.cn/dataset-detail/CommonSenseQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CommonSenseQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CommonSenseQA", "title": "CommonSenseQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1336-compbench", "domain": "多模态", "name": "CompBench", "released": "2024-07-23", "url": "https://hub.opencompass.org.cn/dataset-detail/CompBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CompBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CompBench", "title": "CompBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1679-contextualjudgebench", "domain": "理解", "name": "ContextualJudgeBench", "released": "2025-03-19", "url": "https://hub.opencompass.org.cn/dataset-detail/ContextualJudgeBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ContextualJudgeBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ContextualJudgeBench", "title": "ContextualJudgeBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1269-convbench", "domain": "推理", "name": "ConvBench", "released": "2024-03-29", "url": "https://hub.opencompass.org.cn/dataset-detail/ConvBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ConvBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ConvBench", "title": "ConvBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1643-creation-mmbench", "domain": "多模态", "name": "Creation-MMBench", "released": "2025-03-18", "url": "https://hub.opencompass.org.cn/dataset-detail/Creation-MMBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Creation-MMBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Creation-MMBench", "title": "Creation-MMBench"}, {"benchmarks": [{"benchmark_id": "opencompass-568-criticbench", "domain": "理解", "name": "CriticBench", "released": "2024-02-22", "url": "https://hub.opencompass.org.cn/dataset-detail/CriticBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CriticBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CriticBench", "title": "CriticBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1739-crosswordbench", "domain": "多模态", "name": "CrossWordBench", "released": "2025-03-30", "url": "https://hub.opencompass.org.cn/dataset-detail/CrossWordBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CrossWordBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CrossWordBench", "title": "CrossWordBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1123-crows-pairs", "domain": "安全", "name": "Crows-Pairs", "released": "2020-09-30", "url": "https://hub.opencompass.org.cn/dataset-detail/Crows-Pairs"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Crows-Pairs", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Crows-Pairs", "title": "Crows-Pairs"}, {"benchmarks": [{"benchmark_id": "opencompass-1349-cyberseceval", "domain": "安全", "name": "CyberSecEval", "released": "2023-12-07", "url": "https://hub.opencompass.org.cn/dataset-detail/CyberSecEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CyberSecEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/CyberSecEval", "title": "CyberSecEval"}, {"benchmarks": [{"benchmark_id": "opencompass-2030-dabstep", "domain": "智能体", "name": "DABstep", "released": "2025-06-30", "url": "https://hub.opencompass.org.cn/dataset-detail/DABstep"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DABstep", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DABstep", "title": "DABstep"}, {"benchmarks": [{"benchmark_id": "opencompass-2032-dice-bench", "domain": "智能体", "name": "DICE-BENCH", "released": "2025-06-28", "url": "https://hub.opencompass.org.cn/dataset-detail/DICE-BENCH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DICE-BENCH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DICE-BENCH", "title": "DICE-BENCH"}, {"benchmarks": [{"benchmark_id": "opencompass-1648-dme", "domain": "多模态", "name": "DME", "released": "2024-10-11", "url": "https://hub.opencompass.org.cn/dataset-detail/DME"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DME", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DME", "title": "DME"}, {"benchmarks": [{"benchmark_id": "opencompass-1731-dove", "domain": "其他", "name": "DOVE", "released": "2025-03-04", "url": "https://hub.opencompass.org.cn/dataset-detail/DOVE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DOVE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DOVE", "title": "DOVE"}, {"benchmarks": [{"benchmark_id": "opencompass-2045-dragon", "domain": "知识", "name": "DRAGON", "released": "2025-07-08", "url": "https://hub.opencompass.org.cn/dataset-detail/DRAGON"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DRAGON", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DRAGON", "title": "DRAGON"}, {"benchmarks": [{"benchmark_id": "opencompass-536-drop", "domain": "推理", "name": "DROP", "released": "2019-03-01", "url": "https://hub.opencompass.org.cn/dataset-detail/DROP"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DROP", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DROP", "title": "DROP"}, {"benchmarks": [{"benchmark_id": "opencompass-544-ds-1000", "domain": "代码", "name": "DS-1000", "released": "2022-11-18", "url": "https://hub.opencompass.org.cn/dataset-detail/DS-1000"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DS-1000", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DS-1000", "title": "DS-1000"}, {"benchmarks": [{"benchmark_id": "opencompass-1078-debugbench", "domain": "代码", "name": "DebugBench", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/DebugBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DebugBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DebugBench", "title": "DebugBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1987-deepresearchbench", "domain": "智能体", "name": "DeepResearchBench", "released": "2025-06-13", "url": "https://hub.opencompass.org.cn/dataset-detail/DeepResearchBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DeepResearchBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DeepResearchBench", "title": "DeepResearchBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1605-deepfake-eval-2024", "domain": "其他", "name": "Deepfake-Eval-2024", "released": "2025-03-04", "url": "https://hub.opencompass.org.cn/dataset-detail/Deepfake-Eval-2024"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Deepfake-Eval-2024", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Deepfake-Eval-2024", "title": "Deepfake-Eval-2024"}, {"benchmarks": [{"benchmark_id": "opencompass-1971-dycodeeval", "domain": "强推理", "name": "DyCodeEval", "released": "2025-06-23", "url": "https://hub.opencompass.org.cn/dataset-detail/DyCodeEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DyCodeEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DyCodeEval", "title": "DyCodeEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1374-dynamath", "domain": "多模态", "name": "DynaMath", "released": "2024-10-29", "url": "https://hub.opencompass.org.cn/dataset-detail/DynaMath"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DynaMath", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/DynaMath", "title": "DynaMath"}, {"benchmarks": [{"benchmark_id": "opencompass-1080-e-eval", "domain": "学科", "name": "E-EVAL", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/E-EVAL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/E-EVAL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/E-EVAL", "title": "E-EVAL"}, {"benchmarks": [{"benchmark_id": "opencompass-1327-ehrnoteqa", "domain": "知识", "name": "EHRNoteQA", "released": "2024-02-25", "url": "https://hub.opencompass.org.cn/dataset-detail/EHRNoteQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EHRNoteQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EHRNoteQA", "title": "EHRNoteQA"}, {"benchmarks": [{"benchmark_id": "opencompass-2571-elbench", "domain": "推理", "name": "ELBench", "released": "2026-06-23", "url": "https://hub.opencompass.org.cn/dataset-detail/ELBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ELBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ELBench", "title": "ELBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1640-emma", "domain": "多模态", "name": "EMMA", "released": "2025-01-09", "url": "https://hub.opencompass.org.cn/dataset-detail/EMMA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EMMA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EMMA", "title": "EMMA"}, {"benchmarks": [{"benchmark_id": "opencompass-522-eprstmt", "domain": "理解", "name": "EPRSTMT", "released": "2021-07-15", "url": "https://hub.opencompass.org.cn/dataset-detail/EPRSTMT"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EPRSTMT", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EPRSTMT", "title": "EPRSTMT"}, {"benchmarks": [{"benchmark_id": "opencompass-1892-er-reason", "domain": "推理", "name": "ER-Reason", "released": "2025-05-28", "url": "https://hub.opencompass.org.cn/dataset-detail/ER-Reason"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ER-Reason", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ER-Reason", "title": "ER-Reason"}, {"benchmarks": [{"benchmark_id": "opencompass-1843-ewmbench", "domain": "多模态", "name": "EWMBench", "released": "2025-05-16", "url": "https://hub.opencompass.org.cn/dataset-detail/EWMBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EWMBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EWMBench", "title": "EWMBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1970-editinspector", "domain": "创作", "name": "EditInspector", "released": "2025-06-11", "url": "https://hub.opencompass.org.cn/dataset-detail/EditInspector"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EditInspector", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EditInspector", "title": "EditInspector"}, {"benchmarks": [{"benchmark_id": "opencompass-1604-egonormia", "domain": "学科", "name": "EgoNormia", "released": "2025-02-27", "url": "https://hub.opencompass.org.cn/dataset-detail/EgoNormia"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EgoNormia", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EgoNormia", "title": "EgoNormia"}, {"benchmarks": [{"benchmark_id": "opencompass-1243-embodiedagentinterface", "domain": "推理", "name": "EmbodiedAgentInterface", "released": "2024-06-24", "url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedAgentInterface"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EmbodiedAgentInterface", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedAgentInterface", "title": "EmbodiedAgentInterface"}, {"benchmarks": [{"benchmark_id": "opencompass-1564-embodiedbench", "domain": "多模态", "name": "EmbodiedBench", "released": "2025-02-13", "url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EmbodiedBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedBench", "title": "EmbodiedBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1733-feabench", "domain": "智能体", "name": "FEABench", "released": "2025-04-08", "url": "https://hub.opencompass.org.cn/dataset-detail/FEABench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FEABench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FEABench", "title": "FEABench"}, {"benchmarks": [{"benchmark_id": "opencompass-1324-flub", "domain": "推理", "name": "FLUB", "released": "2024-02-16", "url": "https://hub.opencompass.org.cn/dataset-detail/FLUB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FLUB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FLUB", "title": "FLUB"}, {"benchmarks": [{"benchmark_id": "opencompass-1145-freb-tqa", "domain": "理解", "name": "FREB-TQA", "released": "2024-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/FREB-TQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FREB-TQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FREB-TQA", "title": "FREB-TQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1980-falsereject", "domain": "推理", "name": "FalseReject", "released": "2025-05-12", "url": "https://hub.opencompass.org.cn/dataset-detail/FalseReject"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FalseReject", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FalseReject", "title": "FalseReject"}, {"benchmarks": [{"benchmark_id": "opencompass-1618-fedmabench", "domain": "智能体", "name": "FedMABench", "released": "2025-03-07", "url": "https://hub.opencompass.org.cn/dataset-detail/FedMABench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FedMABench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FedMABench", "title": "FedMABench"}, {"benchmarks": [{"benchmark_id": "opencompass-895-fin-eva", "domain": "知识", "name": "Fin-Eva", "released": "2023-12-21", "url": "https://hub.opencompass.org.cn/dataset-detail/Fin-Eva"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Fin-Eva", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Fin-Eva", "title": "Fin-Eva"}, {"benchmarks": [{"benchmark_id": "opencompass-945-flames", "domain": "安全", "name": "Flames", "released": "2024-03-13", "url": "https://hub.opencompass.org.cn/dataset-detail/Flames"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Flames", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Flames", "title": "Flames"}, {"benchmarks": [{"benchmark_id": "opencompass-509-flores", "domain": "语言", "name": "Flores", "released": "2021-06-06", "url": "https://hub.opencompass.org.cn/dataset-detail/Flores"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Flores", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Flores", "title": "Flores"}, {"benchmarks": [{"benchmark_id": "opencompass-1671-forensics-bench", "domain": "多模态", "name": "Forensics-bench", "released": "2025-03-24", "url": "https://hub.opencompass.org.cn/dataset-detail/Forensics-bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Forensics-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Forensics-bench", "title": "Forensics-bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1737-fortisavqa", "domain": "多模态", "name": "FortisAVQA", "released": "2025-04-02", "url": "https://hub.opencompass.org.cn/dataset-detail/FortisAVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FortisAVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/FortisAVQA", "title": "FortisAVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1599-gaia", "domain": "智能体", "name": "GAIA", "released": "2023-11-21", "url": "https://hub.opencompass.org.cn/dataset-detail/GAIA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAIA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAIA", "title": "GAIA"}, {"benchmarks": [{"benchmark_id": "opencompass-500-gaokao-bench", "domain": "学科", "name": "GAOKAO-Bench", "released": "2023-05-21", "url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAOKAO-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-Bench", "title": "GAOKAO-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1083-gaokao-mm", "domain": "学科", "name": "GAOKAO-MM", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-MM"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAOKAO-MM", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-MM", "title": "GAOKAO-MM"}, {"benchmarks": [{"benchmark_id": "opencompass-2574-gauge", "domain": "其他", "name": "GAUGE", "released": "2026-08-12", "url": "https://hub.opencompass.org.cn/dataset-detail/GAUGE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAUGE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GAUGE", "title": "GAUGE"}, {"benchmarks": [{"benchmark_id": "opencompass-1927-gmaimmbench", "domain": "多模态", "name": "GMAIMMBench", "released": "2024-08-31", "url": "https://hub.opencompass.org.cn/dataset-detail/GMAIMMBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GMAIMMBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GMAIMMBench", "title": "GMAIMMBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1135-gpqa", "domain": "知识", "name": "GPQA", "released": "2023-11-20", "url": "https://hub.opencompass.org.cn/dataset-detail/GPQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GPQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GPQA", "title": "GPQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1740-gpt-imgeval", "domain": "多模态", "name": "GPT-ImgEval", "released": "2025-04-03", "url": "https://hub.opencompass.org.cn/dataset-detail/GPT-ImgEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GPT-ImgEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GPT-ImgEval", "title": "GPT-ImgEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1272-gsm1k", "domain": "数学", "name": "GSM1k", "released": "2024-05-01", "url": "https://hub.opencompass.org.cn/dataset-detail/GSM1k"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GSM1k", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GSM1k", "title": "GSM1k"}, {"benchmarks": [{"benchmark_id": "opencompass-535-gsm8k", "domain": "数学", "name": "GSM8K", "released": "2021-04-01", "url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GSM8K", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K", "title": "GSM8K"}, {"benchmarks": [{"benchmark_id": "opencompass-2233-gsm8k-v", "domain": "多模态", "name": "GSM8K-V", "released": "2025-09-28", "url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K-V"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GSM8K-V", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K-V", "title": "GSM8K-V"}, {"benchmarks": [{"benchmark_id": "opencompass-1328-gta", "domain": "智能体", "name": "GTA", "released": "2024-07-11", "url": "https://hub.opencompass.org.cn/dataset-detail/GTA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GTA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GTA", "title": "GTA"}, {"benchmarks": [{"benchmark_id": "opencompass-1376-genai-bench", "domain": "多模态", "name": "GenAI-Bench", "released": "2024-06-19", "url": "https://hub.opencompass.org.cn/dataset-detail/GenAI-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GenAI-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GenAI-Bench", "title": "GenAI-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2081-general-bench", "domain": "多模态", "name": "General-Bench", "released": "2025-05-07", "url": "https://hub.opencompass.org.cn/dataset-detail/General-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/General-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/General-Bench", "title": "General-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1865-gitgoodbench", "domain": "代码", "name": "GitGoodBench", "released": "2025-05-28", "url": "https://hub.opencompass.org.cn/dataset-detail/GitGoodBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GitGoodBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GitGoodBench", "title": "GitGoodBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1128-gorilla", "domain": "智能体", "name": "Gorilla", "released": "2023-05-24", "url": "https://hub.opencompass.org.cn/dataset-detail/Gorilla"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Gorilla", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Gorilla", "title": "Gorilla"}, {"benchmarks": [{"benchmark_id": "opencompass-1098-grailqa", "domain": "知识", "name": "GrailQA", "released": "2021-02-22", "url": "https://hub.opencompass.org.cn/dataset-detail/GrailQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GrailQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GrailQA", "title": "GrailQA"}, {"benchmarks": [{"benchmark_id": "opencompass-2027-groundingsuite", "domain": "多模态", "name": "GroundingSuite", "released": "2025-07-05", "url": "https://hub.opencompass.org.cn/dataset-detail/GroundingSuite"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GroundingSuite", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/GroundingSuite", "title": "GroundingSuite"}, {"benchmarks": [{"benchmark_id": "opencompass-2035-gym4real", "domain": "其他", "name": "Gym4ReaL", "released": "2025-06-30", "url": "https://hub.opencompass.org.cn/dataset-detail/Gym4ReaL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Gym4ReaL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Gym4ReaL", "title": "Gym4ReaL"}, {"benchmarks": [{"benchmark_id": "opencompass-1842-hardmath2", "domain": "推理", "name": "HARDMath2", "released": "2025-05-17", "url": "https://hub.opencompass.org.cn/dataset-detail/HARDMath2"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HARDMath2", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HARDMath2", "title": "HARDMath2"}, {"benchmarks": [{"benchmark_id": "opencompass-2031-herb", "domain": "其他", "name": "HERB", "released": "2025-06-29", "url": "https://hub.opencompass.org.cn/dataset-detail/HERB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HERB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HERB", "title": "HERB"}, {"benchmarks": [{"benchmark_id": "opencompass-1355-hallusionbench", "domain": "多模态", "name": "HallusionBench", "released": "2023-10-23", "url": "https://hub.opencompass.org.cn/dataset-detail/HallusionBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HallusionBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HallusionBench", "title": "HallusionBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1121-halueval", "domain": "安全", "name": "HaluEval", "released": "2023-10-23", "url": "https://hub.opencompass.org.cn/dataset-detail/HaluEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HaluEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HaluEval", "title": "HaluEval"}, {"benchmarks": [{"benchmark_id": "opencompass-531-hellaswag", "domain": "推理", "name": "HellaSwag", "released": "2019-05-19", "url": "https://hub.opencompass.org.cn/dataset-detail/HellaSwag"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HellaSwag", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HellaSwag", "title": "HellaSwag"}, {"benchmarks": [{"benchmark_id": "opencompass-1097-hellobench", "domain": "长文本", "name": "HelloBench", "released": "2024-09-24", "url": "https://hub.opencompass.org.cn/dataset-detail/HelloBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HelloBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HelloBench", "title": "HelloBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1581-holobench", "domain": "推理", "name": "HoloBench", "released": "2024-10-15", "url": "https://hub.opencompass.org.cn/dataset-detail/HoloBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HoloBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HoloBench", "title": "HoloBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1108-hotpotqa", "domain": "强推理", "name": "HotpotQA", "released": "2018-09-25", "url": "https://hub.opencompass.org.cn/dataset-detail/HotpotQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HotpotQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HotpotQA", "title": "HotpotQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1992-htfllib", "domain": "安全", "name": "HtFLlib", "released": "2025-06-04", "url": "https://hub.opencompass.org.cn/dataset-detail/HtFLlib"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HtFLlib", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HtFLlib", "title": "HtFLlib"}, {"benchmarks": [{"benchmark_id": "opencompass-537-humaneval", "domain": "代码", "name": "HumanEval", "released": "2021-07-08", "url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HumanEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval", "title": "HumanEval"}, {"benchmarks": [{"benchmark_id": "opencompass-543-humaneval-x", "domain": "代码", "name": "HumanEval-X", "released": "2023-03-30", "url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval-X"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HumanEval-X", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval-X", "title": "HumanEval-X"}, {"benchmarks": [{"benchmark_id": "opencompass-1782-hypobench", "domain": "其他", "name": "HypoBench", "released": "2025-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/HypoBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HypoBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HypoBench", "title": "HypoBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1786-hypoeval", "domain": "其他", "name": "HypoEval", "released": "2025-04-09", "url": "https://hub.opencompass.org.cn/dataset-detail/HypoEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HypoEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/HypoEval", "title": "HypoEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1136-ifeval", "domain": "指令跟随", "name": "IFEval", "released": "2023-11-14", "url": "https://hub.opencompass.org.cn/dataset-detail/IFEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IFEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IFEval", "title": "IFEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1617-ifir", "domain": "指令跟随", "name": "IFIR", "released": "2025-03-06", "url": "https://hub.opencompass.org.cn/dataset-detail/IFIR"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IFIR", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IFIR", "title": "IFIR"}, {"benchmarks": [{"benchmark_id": "opencompass-1274-imdl-benco", "domain": "多模态", "name": "IMDL-BenCo", "released": "2024-06-15", "url": "https://hub.opencompass.org.cn/dataset-detail/IMDL-BenCo"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IMDL-BenCo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IMDL-BenCo", "title": "IMDL-BenCo"}, {"benchmarks": [{"benchmark_id": "opencompass-1840-iqbench", "domain": "多模态", "name": "IQBench", "released": "2025-05-17", "url": "https://hub.opencompass.org.cn/dataset-detail/IQBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IQBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IQBench", "title": "IQBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2091-is-bench", "domain": "多模态", "name": "IS-Bench", "released": "2025-07-07", "url": "https://hub.opencompass.org.cn/dataset-detail/IS-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IS-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IS-Bench", "title": "IS-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2077-itbench", "domain": "代码", "name": "ITBench", "released": "2025-02-07", "url": "https://hub.opencompass.org.cn/dataset-detail/ITBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ITBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ITBench", "title": "ITBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1320-iac-eval", "domain": "代码", "name": "IaC-Eval", "released": "2024-09-26", "url": "https://hub.opencompass.org.cn/dataset-detail/IaC-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IaC-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IaC-Eval", "title": "IaC-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-1670-indicmmlu-pro", "domain": "语言", "name": "IndicMMLU-Pro", "released": "2025-01-28", "url": "https://hub.opencompass.org.cn/dataset-detail/IndicMMLU-Pro"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IndicMMLU-Pro", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IndicMMLU-Pro", "title": "IndicMMLU-Pro"}, {"benchmarks": [{"benchmark_id": "opencompass-1322-infibench", "domain": "代码", "name": "InfiBench", "released": "2024-03-11", "url": "https://hub.opencompass.org.cn/dataset-detail/InfiBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InfiBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InfiBench", "title": "InfiBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1085-infobench", "domain": "指令跟随", "name": "InfoBench", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/InfoBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InfoBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InfoBench", "title": "InfoBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1149-instrusum", "domain": "理解", "name": "InstruSum", "released": "2024-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/InstruSum"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InstruSum", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InstruSum", "title": "InstruSum"}, {"benchmarks": [{"benchmark_id": "opencompass-1976-intphys2", "domain": "理解", "name": "IntPhys2", "released": "2025-06-11", "url": "https://hub.opencompass.org.cn/dataset-detail/IntPhys2"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IntPhys2", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/IntPhys2", "title": "IntPhys2"}, {"benchmarks": [{"benchmark_id": "opencompass-2151-interndata-a1", "domain": "多模态", "name": "InternData-A1", "released": "2025-07-26", "url": "https://hub.opencompass.org.cn/dataset-detail/InternData-A1"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InternData-A1", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InternData-A1", "title": "InternData-A1"}, {"benchmarks": [{"benchmark_id": "opencompass-2153-interndata-m1", "domain": "多模态", "name": "InternData-M1", "released": "2025-07-26", "url": "https://hub.opencompass.org.cn/dataset-detail/InternData-M1"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InternData-M1", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InternData-M1", "title": "InternData-M1"}, {"benchmarks": [{"benchmark_id": "opencompass-2152-interndata-n1", "domain": "多模态", "name": "InternData-N1", "released": "2025-07-26", "url": "https://hub.opencompass.org.cn/dataset-detail/InternData-N1"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InternData-N1", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/InternData-N1", "title": "InternData-N1"}, {"benchmarks": [{"benchmark_id": "opencompass-2073-j1-bench", "domain": "智能体", "name": "J1-Bench", "released": "2025-07-17", "url": "https://hub.opencompass.org.cn/dataset-detail/J1-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/J1-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/J1-Bench", "title": "J1-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1554-jl1-cd", "domain": "其他", "name": "JL1-CD", "released": "2025-02-19", "url": "https://hub.opencompass.org.cn/dataset-detail/JL1-CD"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JL1-CD", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JL1-CD", "title": "JL1-CD"}, {"benchmarks": [{"benchmark_id": "opencompass-2130-job-complex", "domain": "其他", "name": "JOB-Complex", "released": "2025-07-10", "url": "https://hub.opencompass.org.cn/dataset-detail/JOB-Complex"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JOB-Complex", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JOB-Complex", "title": "JOB-Complex"}, {"benchmarks": [{"benchmark_id": "opencompass-1337-jailtrickbench", "domain": "安全", "name": "JailTrickBench", "released": "2024-06-13", "url": "https://hub.opencompass.org.cn/dataset-detail/JailTrickBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JailTrickBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JailTrickBench", "title": "JailTrickBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1583-judgebench", "domain": "其他", "name": "JudgeBench", "released": "2024-10-16", "url": "https://hub.opencompass.org.cn/dataset-detail/JudgeBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JudgeBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/JudgeBench", "title": "JudgeBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2076-k-sort-arena", "domain": "多模态", "name": "K-Sort-Arena", "released": "2024-09-02", "url": "https://hub.opencompass.org.cn/dataset-detail/K-Sort-Arena"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/K-Sort-Arena", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/K-Sort-Arena", "title": "K-Sort-Arena"}, {"benchmarks": [{"benchmark_id": "opencompass-1543-kitab-bench", "domain": "多模态", "name": "KITAB-Bench", "released": "2025-02-20", "url": "https://hub.opencompass.org.cn/dataset-detail/KITAB-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KITAB-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KITAB-Bench", "title": "KITAB-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2126-kmmlu-redux", "domain": "知识", "name": "KMMLU-Redux", "released": "2025-07-11", "url": "https://hub.opencompass.org.cn/dataset-detail/KMMLU-Redux"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KMMLU-Redux", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KMMLU-Redux", "title": "KMMLU-Redux"}, {"benchmarks": [{"benchmark_id": "opencompass-1706-koffvqa", "domain": "多模态", "name": "KOFFVQA", "released": "2025-03-31", "url": "https://hub.opencompass.org.cn/dataset-detail/KOFFVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KOFFVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KOFFVQA", "title": "KOFFVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1245-kor-bench", "domain": "强推理", "name": "KOR-Bench", "released": "2024-10-09", "url": "https://hub.opencompass.org.cn/dataset-detail/KOR-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KOR-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KOR-Bench", "title": "KOR-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1635-knowlogic", "domain": "推理", "name": "KnowLogic", "released": "2025-03-08", "url": "https://hub.opencompass.org.cn/dataset-detail/KnowLogic"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KnowLogic", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/KnowLogic", "title": "KnowLogic"}, {"benchmarks": [{"benchmark_id": "opencompass-541-l-eval", "domain": "长文本", "name": "L-Eval", "released": "2023-10-04", "url": "https://hub.opencompass.org.cn/dataset-detail/L-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/L-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/L-Eval", "title": "L-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-523-lambada", "domain": "理解", "name": "LAMBADA", "released": "2016-06-20", "url": "https://hub.opencompass.org.cn/dataset-detail/LAMBADA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LAMBADA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LAMBADA", "title": "LAMBADA"}, {"benchmarks": [{"benchmark_id": "opencompass-520-lcsts", "domain": "理解", "name": "LCSTS", "released": "2015-06-19", "url": "https://hub.opencompass.org.cn/dataset-detail/LCSTS"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LCSTS", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LCSTS", "title": "LCSTS"}, {"benchmarks": [{"benchmark_id": "opencompass-2422-lens", "domain": "多模态", "name": "LENS", "released": "2025-06-01", "url": "https://hub.opencompass.org.cn/dataset-detail/LENS"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LENS", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LENS", "title": "LENS"}, {"benchmarks": [{"benchmark_id": "opencompass-1238-lingoly", "domain": "推理", "name": "LINGOLY", "released": "2024-06-10", "url": "https://hub.opencompass.org.cn/dataset-detail/LINGOLY"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LINGOLY", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LINGOLY", "title": "LINGOLY"}, {"benchmarks": [{"benchmark_id": "opencompass-1838-llm-babybench", "domain": "强推理", "name": "LLM-BabyBench", "released": "2025-05-17", "url": "https://hub.opencompass.org.cn/dataset-detail/LLM-BabyBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLM-BabyBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLM-BabyBench", "title": "LLM-BabyBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1752-llm-srbench", "domain": "推理", "name": "LLM-SRBench", "released": "2025-04-14", "url": "https://hub.opencompass.org.cn/dataset-detail/LLM-SRBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLM-SRBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLM-SRBench", "title": "LLM-SRBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1325-llm-uncertainty-bench", "domain": "其他", "name": "LLM-Uncertainty-Bench", "released": "2024-01-23", "url": "https://hub.opencompass.org.cn/dataset-detail/LLM-Uncertainty-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLM-Uncertainty-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLM-Uncertainty-Bench", "title": "LLM-Uncertainty-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2047-llmthinkbench", "domain": "推理", "name": "LLMThinkBench", "released": "2025-07-05", "url": "https://hub.opencompass.org.cn/dataset-detail/LLMThinkBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLMThinkBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLMThinkBench", "title": "LLMThinkBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1360-llava-bench", "domain": "多模态", "name": "LLaVA-Bench", "released": "2023-04-17", "url": "https://hub.opencompass.org.cn/dataset-detail/LLaVA-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLaVA-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LLaVA-Bench", "title": "LLaVA-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2088-lmact", "domain": "长文本", "name": "LMAct", "released": "2024-12-02", "url": "https://hub.opencompass.org.cn/dataset-detail/LMAct"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LMAct", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LMAct", "title": "LMAct"}, {"benchmarks": [{"benchmark_id": "opencompass-1572-loki", "domain": "多模态", "name": "LOKI", "released": "2024-10-13", "url": "https://hub.opencompass.org.cn/dataset-detail/LOKI"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LOKI", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LOKI", "title": "LOKI"}, {"benchmarks": [{"benchmark_id": "opencompass-1335-ltmbenchmark", "domain": "智能体", "name": "LTMbenchmark", "released": "2024-09-30", "url": "https://hub.opencompass.org.cn/dataset-detail/LTMbenchmark"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LTMbenchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LTMbenchmark", "title": "LTMbenchmark"}, {"benchmarks": [{"benchmark_id": "opencompass-564-lv-eval", "domain": "长文本", "name": "LV-Eval", "released": "2024-02-18", "url": "https://hub.opencompass.org.cn/dataset-detail/LV-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LV-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LV-Eval", "title": "LV-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-1924-lamp-qa", "domain": "其他", "name": "LaMP-QA", "released": "2025-05-30", "url": "https://hub.opencompass.org.cn/dataset-detail/LaMP-QA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LaMP-QA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LaMP-QA", "title": "LaMP-QA"}, {"benchmarks": [{"benchmark_id": "opencompass-2089-lara", "domain": "长文本", "name": "LaRA", "released": "2025-02-14", "url": "https://hub.opencompass.org.cn/dataset-detail/LaRA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LaRA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LaRA", "title": "LaRA"}, {"benchmarks": [{"benchmark_id": "opencompass-2129-langnavbench", "domain": "理解", "name": "LangNavBench", "released": "2025-07-09", "url": "https://hub.opencompass.org.cn/dataset-detail/LangNavBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LangNavBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LangNavBench", "title": "LangNavBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1246-livebench", "domain": "知识", "name": "LiveBench", "released": "2024-06-12", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveBench", "title": "LiveBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1413-livecodebench", "domain": "代码", "name": "LiveCodeBench", "released": "2024-03-12", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveCodeBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveCodeBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveCodeBench", "title": "LiveCodeBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1787-livelongbench", "domain": "推理", "name": "LiveLongBench", "released": "2025-04-24", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveLongBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveLongBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveLongBench", "title": "LiveLongBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1397-livemathbench", "domain": "强推理", "name": "LiveMathBench", "released": "2024-12-17", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveMathBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveMathBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LiveMathBench", "title": "LiveMathBench"}, {"benchmarks": [{"benchmark_id": "opencompass-542-longbench", "domain": "长文本", "name": "LongBench", "released": "2023-08-28", "url": "https://hub.opencompass.org.cn/dataset-detail/LongBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LongBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LongBench", "title": "LongBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2067-longvale", "domain": "多模态", "name": "LongVALE", "released": "2025-04-17", "url": "https://hub.opencompass.org.cn/dataset-detail/LongVALE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LongVALE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LongVALE", "title": "LongVALE"}, {"benchmarks": [{"benchmark_id": "opencompass-1510-longvideobench", "domain": "多模态", "name": "LongVideoBench", "released": "2024-07-22", "url": "https://hub.opencompass.org.cn/dataset-detail/LongVideoBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LongVideoBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LongVideoBench", "title": "LongVideoBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1944-loopnav", "domain": "多模态", "name": "LoopNav", "released": "2025-05-15", "url": "https://hub.opencompass.org.cn/dataset-detail/LoopNav"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LoopNav", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/LoopNav", "title": "LoopNav"}, {"benchmarks": [{"benchmark_id": "opencompass-1147-m3t", "domain": "理解", "name": "M3T", "released": "2024-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/M3T"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/M3T", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/M3T", "title": "M3T"}, {"benchmarks": [{"benchmark_id": "opencompass-1609-mask", "domain": "其他", "name": "MASK", "released": "2025-03-05", "url": "https://hub.opencompass.org.cn/dataset-detail/MASK"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MASK", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MASK", "title": "MASK"}, {"benchmarks": [{"benchmark_id": "opencompass-534-math", "domain": "数学", "name": "MATH", "released": "2021-11-08", "url": "https://hub.opencompass.org.cn/dataset-detail/MATH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MATH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MATH", "title": "MATH"}, {"benchmarks": [{"benchmark_id": "opencompass-1848-mavos-dd", "domain": "多模态", "name": "MAVOS-DD", "released": "2025-05-16", "url": "https://hub.opencompass.org.cn/dataset-detail/MAVOS-DD"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MAVOS-DD", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MAVOS-DD", "title": "MAVOS-DD"}, {"benchmarks": [{"benchmark_id": "opencompass-538-mbpp", "domain": "代码", "name": "MBPP", "released": "2021-08-16", "url": "https://hub.opencompass.org.cn/dataset-detail/MBPP"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MBPP", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MBPP", "title": "MBPP"}, {"benchmarks": [{"benchmark_id": "opencompass-1606-mcitebench", "domain": "多模态", "name": "MCiteBench", "released": "2025-03-05", "url": "https://hub.opencompass.org.cn/dataset-detail/MCiteBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MCiteBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MCiteBench", "title": "MCiteBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1893-medal", "domain": "其他", "name": "MEDAL", "released": "2025-05-28", "url": "https://hub.opencompass.org.cn/dataset-detail/MEDAL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MEDAL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MEDAL", "title": "MEDAL"}, {"benchmarks": [{"benchmark_id": "opencompass-2079-mer-unibench", "domain": "多模态", "name": "MER-UniBench", "released": "2025-01-27", "url": "https://hub.opencompass.org.cn/dataset-detail/MER-UniBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MER-UniBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MER-UniBench", "title": "MER-UniBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2085-mib", "domain": "理解", "name": "MIB", "released": "2025-04-17", "url": "https://hub.opencompass.org.cn/dataset-detail/MIB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIB", "title": "MIB"}, {"benchmarks": [{"benchmark_id": "opencompass-1781-mieb", "domain": "多模态", "name": "MIEB", "released": "2025-04-14", "url": "https://hub.opencompass.org.cn/dataset-detail/MIEB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIEB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIEB", "title": "MIEB"}, {"benchmarks": [{"benchmark_id": "opencompass-1844-miracl-vision", "domain": "多模态", "name": "MIRACL-VISION", "released": "2025-05-22", "url": "https://hub.opencompass.org.cn/dataset-detail/MIRACL-VISION"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIRACL-VISION", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIRACL-VISION", "title": "MIRACL-VISION"}, {"benchmarks": [{"benchmark_id": "opencompass-1142-mirage", "domain": "其他", "name": "MIRAGE", "released": "2024-08-16", "url": "https://hub.opencompass.org.cn/dataset-detail/MIRAGE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIRAGE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MIRAGE", "title": "MIRAGE"}, {"benchmarks": [{"benchmark_id": "opencompass-1623-mj-bench", "domain": "多模态", "name": "MJ-Bench", "released": "2024-07-05", "url": "https://hub.opencompass.org.cn/dataset-detail/MJ-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MJ-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MJ-Bench", "title": "MJ-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1099-mkqa", "domain": "知识", "name": "MKQA", "released": "2021-08-17", "url": "https://hub.opencompass.org.cn/dataset-detail/MKQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MKQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MKQA", "title": "MKQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1779-mlrc-bench", "domain": "智能体", "name": "MLRC-Bench", "released": "2025-04-13", "url": "https://hub.opencompass.org.cn/dataset-detail/MLRC-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MLRC-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MLRC-Bench", "title": "MLRC-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1512-mlvu", "domain": "多模态", "name": "MLVU", "released": "2024-06-06", "url": "https://hub.opencompass.org.cn/dataset-detail/MLVU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MLVU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MLVU", "title": "MLVU"}, {"benchmarks": [{"benchmark_id": "opencompass-1558-mm-alignbench", "domain": "多模态", "name": "MM-AlignBench", "released": "2025-02-25", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-AlignBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-AlignBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-AlignBench", "title": "MM-AlignBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1566-mm-iq", "domain": "多模态", "name": "MM-IQ", "released": "2025-02-02", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-IQ"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-IQ", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-IQ", "title": "MM-IQ"}, {"benchmarks": [{"benchmark_id": "opencompass-1539-mm-rlhf", "domain": "多模态", "name": "MM-RLHF", "released": "2025-02-04", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-RLHF"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-RLHF", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-RLHF", "title": "MM-RLHF"}, {"benchmarks": [{"benchmark_id": "opencompass-1356-mm-vet", "domain": "多模态", "name": "MM-Vet", "released": "2023-08-04", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-Vet"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-Vet", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MM-Vet", "title": "MM-Vet"}, {"benchmarks": [{"benchmark_id": "opencompass-1585-mmad", "domain": "多模态", "name": "MMAD", "released": "2024-10-12", "url": "https://hub.opencompass.org.cn/dataset-detail/MMAD"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMAD", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMAD", "title": "MMAD"}, {"benchmarks": [{"benchmark_id": "opencompass-1910-mmar", "domain": "强推理", "name": "MMAR", "released": "2025-05-11", "url": "https://hub.opencompass.org.cn/dataset-detail/MMAR"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMAR", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMAR", "title": "MMAR"}, {"benchmarks": [{"benchmark_id": "opencompass-1206-mmbench", "domain": "多模态", "name": "MMBench", "released": "2023-07-12", "url": "https://hub.opencompass.org.cn/dataset-detail/MMBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMBench", "title": "MMBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1207-mmbench-video", "domain": "多模态", "name": "MMBench-Video", "released": "2024-06-20", "url": "https://hub.opencompass.org.cn/dataset-detail/MMBench-Video"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMBench-Video", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMBench-Video", "title": "MMBench-Video"}, {"benchmarks": [{"benchmark_id": "opencompass-1334-mmdu", "domain": "多模态", "name": "MMDU", "released": "2024-06-17", "url": "https://hub.opencompass.org.cn/dataset-detail/MMDU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMDU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMDU", "title": "MMDU"}, {"benchmarks": [{"benchmark_id": "opencompass-1357-mme", "domain": "多模态", "name": "MME", "released": "2023-06-23", "url": "https://hub.opencompass.org.cn/dataset-detail/MME"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MME", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MME", "title": "MME"}, {"benchmarks": [{"benchmark_id": "opencompass-1565-mme-cot", "domain": "强推理", "name": "MME-CoT", "released": "2025-02-13", "url": "https://hub.opencompass.org.cn/dataset-detail/MME-CoT"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MME-CoT", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MME-CoT", "title": "MME-CoT"}, {"benchmarks": [{"benchmark_id": "opencompass-1451-mme-realworld", "domain": "多模态", "name": "MME-RealWorld", "released": "2024-08-23", "url": "https://hub.opencompass.org.cn/dataset-detail/MME-RealWorld"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MME-RealWorld", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MME-RealWorld", "title": "MME-RealWorld"}, {"benchmarks": [{"benchmark_id": "opencompass-1523-mmie", "domain": "多模态", "name": "MMIE", "released": "2024-10-15", "url": "https://hub.opencompass.org.cn/dataset-detail/MMIE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMIE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMIE", "title": "MMIE"}, {"benchmarks": [{"benchmark_id": "opencompass-1547-mmir", "domain": "多模态", "name": "MMIR", "released": "2025-02-22", "url": "https://hub.opencompass.org.cn/dataset-detail/MMIR"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMIR", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMIR", "title": "MMIR"}, {"benchmarks": [{"benchmark_id": "opencompass-1453-mmiu", "domain": "多模态", "name": "MMIU", "released": "2024-08-05", "url": "https://hub.opencompass.org.cn/dataset-detail/MMIU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMIU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMIU", "title": "MMIU"}, {"benchmarks": [{"benchmark_id": "opencompass-1578-mmke-bench", "domain": "其他", "name": "MMKE-Bench", "released": "2025-02-27", "url": "https://hub.opencompass.org.cn/dataset-detail/MMKE-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMKE-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMKE-Bench", "title": "MMKE-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-498-mmlu", "domain": "学科", "name": "MMLU", "released": "2020-09-07", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLU", "title": "MMLU"}, {"benchmarks": [{"benchmark_id": "opencompass-1276-mmlu-pro", "domain": "学科", "name": "MMLU-Pro", "released": "2024-06-03", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLU-Pro"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLU-Pro", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLU-Pro", "title": "MMLU-Pro"}, {"benchmarks": [{"benchmark_id": "opencompass-1851-mmlongbench", "domain": "多模态", "name": "MMLongBench", "released": "2025-05-16", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLongBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench", "title": "MMLongBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1275-mmlongbench-doc", "domain": "长文本", "name": "MMLongBench-Doc", "released": "2024-07-01", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench-Doc"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLongBench-Doc", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench-Doc", "title": "MMLongBench-Doc"}, {"benchmarks": [{"benchmark_id": "opencompass-1248-mmmu", "domain": "知识", "name": "MMMU", "released": "2023-11-27", "url": "https://hub.opencompass.org.cn/dataset-detail/MMMU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMMU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMMU", "title": "MMMU"}, {"benchmarks": [{"benchmark_id": "opencompass-1829-mms-vpr", "domain": "多模态", "name": "MMS-VPR", "released": "2025-05-18", "url": "https://hub.opencompass.org.cn/dataset-detail/MMS-VPR"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMS-VPR", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMS-VPR", "title": "MMS-VPR"}, {"benchmarks": [{"benchmark_id": "opencompass-1922-mmsi-bench", "domain": "多模态", "name": "MMSI-Bench", "released": "2025-05-30", "url": "https://hub.opencompass.org.cn/dataset-detail/MMSI-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMSI-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMSI-Bench", "title": "MMSI-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1584-mmsearch", "domain": "多模态", "name": "MMSearch", "released": "2024-09-19", "url": "https://hub.opencompass.org.cn/dataset-detail/MMSearch"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMSearch", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMSearch", "title": "MMSearch"}, {"benchmarks": [{"benchmark_id": "opencompass-1175-mmstar", "domain": "其他", "name": "MMStar", "released": "2024-04-09", "url": "https://hub.opencompass.org.cn/dataset-detail/MMStar"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMStar", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMStar", "title": "MMStar"}, {"benchmarks": [{"benchmark_id": "opencompass-1364-mmt-bench", "domain": "多模态", "name": "MMT-Bench", "released": "2024-04-24", "url": "https://hub.opencompass.org.cn/dataset-detail/MMT-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMT-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMT-Bench", "title": "MMT-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1736-mmtb", "domain": "智能体", "name": "MMTB", "released": "2025-03-30", "url": "https://hub.opencompass.org.cn/dataset-detail/MMTB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMTB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MMTB", "title": "MMTB"}, {"benchmarks": [{"benchmark_id": "opencompass-1999-morse-500", "domain": "强推理", "name": "MORSE-500", "released": "2025-06-05", "url": "https://hub.opencompass.org.cn/dataset-detail/MORSE-500"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MORSE-500", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MORSE-500", "title": "MORSE-500"}, {"benchmarks": [{"benchmark_id": "opencompass-930-mr-ben-meta-reasoning-benchmark", "domain": "推理", "name": "MR-Ben-Meta-Reasoning-Benchmark", "released": "2024-06-20", "url": "https://hub.opencompass.org.cn/dataset-detail/MR-Ben-Meta-Reasoning-Benchmark"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MR-Ben-Meta-Reasoning-Benchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MR-Ben-Meta-Reasoning-Benchmark", "title": "MR-Ben-Meta-Reasoning-Benchmark"}, {"benchmarks": [{"benchmark_id": "opencompass-1587-mr-gsm8k", "domain": "推理", "name": "MR-GSM8K", "released": "2023-12-28", "url": "https://hub.opencompass.org.cn/dataset-detail/MR-GSM8K"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MR-GSM8K", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MR-GSM8K", "title": "MR-GSM8K"}, {"benchmarks": [{"benchmark_id": "opencompass-1586-mrag-bench", "domain": "多模态", "name": "MRAG-Bench", "released": "2024-10-10", "url": "https://hub.opencompass.org.cn/dataset-detail/MRAG-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MRAG-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MRAG-Bench", "title": "MRAG-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1102-ms-marco", "domain": "知识", "name": "MS_MARCO", "released": "2018-10-31", "url": "https://hub.opencompass.org.cn/dataset-detail/MS_MARCO"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MS_MARCO", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MS_MARCO", "title": "MS_MARCO"}, {"benchmarks": [{"benchmark_id": "opencompass-1172-mt-bench-101", "domain": "理解", "name": "MT-Bench-101", "released": "2024-06-25", "url": "https://hub.opencompass.org.cn/dataset-detail/MT-Bench-101"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MT-Bench-101", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MT-Bench-101", "title": "MT-Bench-101"}, {"benchmarks": [{"benchmark_id": "opencompass-1929-mtcmb", "domain": "推理", "name": "MTCMB", "released": "2025-05-15", "url": "https://hub.opencompass.org.cn/dataset-detail/MTCMB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MTCMB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MTCMB", "title": "MTCMB"}, {"benchmarks": [{"benchmark_id": "opencompass-2034-mteb", "domain": "其他", "name": "MTEB", "released": "2025-06-26", "url": "https://hub.opencompass.org.cn/dataset-detail/MTEB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MTEB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MTEB", "title": "MTEB"}, {"benchmarks": [{"benchmark_id": "opencompass-1502-mtvqa", "domain": "多模态", "name": "MTVQA", "released": "2024-06-04", "url": "https://hub.opencompass.org.cn/dataset-detail/MTVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MTVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MTVQA", "title": "MTVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1509-mvbench", "domain": "多模态", "name": "MVBench", "released": "2023-11-28", "url": "https://hub.opencompass.org.cn/dataset-detail/MVBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MVBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MVBench", "title": "MVBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1537-mvl-sib", "domain": "多模态", "name": "MVL-SIB", "released": "2025-02-18", "url": "https://hub.opencompass.org.cn/dataset-detail/MVL-SIB"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MVL-SIB", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MVL-SIB", "title": "MVL-SIB"}, {"benchmarks": [{"benchmark_id": "opencompass-1905-mvpbench", "domain": "多模态", "name": "MVPBench", "released": "2025-06-02", "url": "https://hub.opencompass.org.cn/dataset-detail/MVPBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MVPBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MVPBench", "title": "MVPBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1694-maritimebench", "domain": "学科", "name": "MaritimeBench", "released": "2025-04-01", "url": "https://hub.opencompass.org.cn/dataset-detail/MaritimeBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MaritimeBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MaritimeBench", "title": "MaritimeBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1847-massive-steps", "domain": "多模态", "name": "Massive-STEPS", "released": "2025-05-16", "url": "https://hub.opencompass.org.cn/dataset-detail/Massive-STEPS"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Massive-STEPS", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Massive-STEPS", "title": "Massive-STEPS"}, {"benchmarks": [{"benchmark_id": "opencompass-1645-mastermindeval", "domain": "推理", "name": "MastermindEval", "released": "2025-03-07", "url": "https://hub.opencompass.org.cn/dataset-detail/MastermindEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MastermindEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MastermindEval", "title": "MastermindEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1089-mathbench", "domain": "数学", "name": "MathBench", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/MathBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathBench", "title": "MathBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1115-mathqa", "domain": "数学", "name": "MathQA", "released": "2019-05-31", "url": "https://hub.opencompass.org.cn/dataset-detail/MathQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathQA", "title": "MathQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1371-mathverse", "domain": "多模态", "name": "MathVerse", "released": "2024-03-21", "url": "https://hub.opencompass.org.cn/dataset-detail/MathVerse"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathVerse", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathVerse", "title": "MathVerse"}, {"benchmarks": [{"benchmark_id": "opencompass-1370-mathvision", "domain": "多模态", "name": "MathVision", "released": "2024-02-22", "url": "https://hub.opencompass.org.cn/dataset-detail/MathVision"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathVision", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathVision", "title": "MathVision"}, {"benchmarks": [{"benchmark_id": "opencompass-1178-mathvista", "domain": "强推理", "name": "MathVista", "released": "2023-10-03", "url": "https://hub.opencompass.org.cn/dataset-detail/MathVista"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathVista", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MathVista", "title": "MathVista"}, {"benchmarks": [{"benchmark_id": "opencompass-1641-medagents-bench", "domain": "推理", "name": "MedAgents-Bench", "released": "2025-03-16", "url": "https://hub.opencompass.org.cn/dataset-detail/MedAgents-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedAgents-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedAgents-Bench", "title": "MedAgents-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1874-medarabiq", "domain": "其他", "name": "MedArabiQ", "released": "2025-05-06", "url": "https://hub.opencompass.org.cn/dataset-detail/MedArabiQ"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedArabiQ", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedArabiQ", "title": "MedArabiQ"}, {"benchmarks": [{"benchmark_id": "opencompass-1287-medbench", "domain": "知识", "name": "MedBench", "released": "2023-12-09", "url": "https://hub.opencompass.org.cn/dataset-detail/MedBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedBench", "title": "MedBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1920-medbookvqa", "domain": "多模态", "name": "MedBookVQA", "released": "2025-05-17", "url": "https://hub.opencompass.org.cn/dataset-detail/MedBookVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedBookVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedBookVQA", "title": "MedBookVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1832-medbrowsecomp", "domain": "其他", "name": "MedBrowseComp", "released": "2025-05-20", "url": "https://hub.opencompass.org.cn/dataset-detail/MedBrowseComp"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedBrowseComp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedBrowseComp", "title": "MedBrowseComp"}, {"benchmarks": [{"benchmark_id": "opencompass-1241-medcalc-bench", "domain": "推理", "name": "MedCalc-Bench", "released": "2024-06-17", "url": "https://hub.opencompass.org.cn/dataset-detail/MedCalc-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedCalc-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedCalc-Bench", "title": "MedCalc-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2417-medhalltune", "domain": "医学", "name": "MedHallTune", "released": "2025-02-28", "url": "https://hub.opencompass.org.cn/dataset-detail/MedHallTune"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedHallTune", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedHallTune", "title": "MedHallTune"}, {"benchmarks": [{"benchmark_id": "opencompass-1548-medhallu", "domain": "其他", "name": "MedHallu", "released": "2025-02-20", "url": "https://hub.opencompass.org.cn/dataset-detail/MedHallu"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedHallu", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedHallu", "title": "MedHallu"}, {"benchmarks": [{"benchmark_id": "opencompass-1326-medjourney", "domain": "推理", "name": "MedJourney", "released": "2024-09-26", "url": "https://hub.opencompass.org.cn/dataset-detail/MedJourney"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedJourney", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedJourney", "title": "MedJourney"}, {"benchmarks": [{"benchmark_id": "opencompass-1331-medsafetybench", "domain": "安全", "name": "MedSafetyBench", "released": "2024-03-06", "url": "https://hub.opencompass.org.cn/dataset-detail/MedSafetyBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedSafetyBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedSafetyBench", "title": "MedSafetyBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1891-medxpertqa", "domain": "学科", "name": "MedXpertQA", "released": "2025-02-09", "url": "https://hub.opencompass.org.cn/dataset-detail/MedXpertQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedXpertQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MedXpertQA", "title": "MedXpertQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1658-milic-eval", "domain": "其他", "name": "MiLiC-Eval", "released": "2025-03-03", "url": "https://hub.opencompass.org.cn/dataset-detail/MiLiC-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MiLiC-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MiLiC-Eval", "title": "MiLiC-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-1667-microvqa", "domain": "多模态", "name": "MicroVQA", "released": "2025-03-17", "url": "https://hub.opencompass.org.cn/dataset-detail/MicroVQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MicroVQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MicroVQA", "title": "MicroVQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1125-mind2web", "domain": "其他", "name": "Mind2Web", "released": "2023-12-09", "url": "https://hub.opencompass.org.cn/dataset-detail/Mind2Web"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Mind2Web", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Mind2Web", "title": "Mind2Web"}, {"benchmarks": [{"benchmark_id": "opencompass-1853-minilongbench", "domain": "理解", "name": "MiniLongBench", "released": "2025-05-15", "url": "https://hub.opencompass.org.cn/dataset-detail/MiniLongBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MiniLongBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MiniLongBench", "title": "MiniLongBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1273-molpuzzle", "domain": "推理", "name": "MolPuzzle", "released": "2024-09-26", "url": "https://hub.opencompass.org.cn/dataset-detail/MolPuzzle"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MolPuzzle", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MolPuzzle", "title": "MolPuzzle"}, {"benchmarks": [{"benchmark_id": "opencompass-1704-mono2stereo", "domain": "其他", "name": "Mono2Stereo", "released": "2025-03-30", "url": "https://hub.opencompass.org.cn/dataset-detail/Mono2Stereo"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Mono2Stereo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Mono2Stereo", "title": "Mono2Stereo"}, {"benchmarks": [{"benchmark_id": "opencompass-1682-motionbench", "domain": "多模态", "name": "MotionBench", "released": "2025-01-06", "url": "https://hub.opencompass.org.cn/dataset-detail/MotionBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MotionBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MotionBench", "title": "MotionBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2154-motionmillion", "domain": "多模态", "name": "MotionMillion", "released": "2025-07-26", "url": "https://hub.opencompass.org.cn/dataset-detail/MotionMillion"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MotionMillion", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MotionMillion", "title": "MotionMillion"}, {"benchmarks": [{"benchmark_id": "opencompass-1751-multiloko", "domain": "语言", "name": "MultiLoKo", "released": "2025-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/MultiLoKo"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MultiLoKo", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/MultiLoKo", "title": "MultiLoKo"}, {"benchmarks": [{"benchmark_id": "opencompass-1783-nppc", "domain": "推理", "name": "NPPC", "released": "2025-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/NPPC"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NPPC", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NPPC", "title": "NPPC"}, {"benchmarks": [{"benchmark_id": "opencompass-513-nq", "domain": "知识", "name": "NQ", "released": "2019-06-03", "url": "https://hub.opencompass.org.cn/dataset-detail/NQ"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NQ", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NQ", "title": "NQ"}, {"benchmarks": [{"benchmark_id": "opencompass-1081-naturalcodebench", "domain": "代码", "name": "NaturalCodeBench", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/NaturalCodeBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NaturalCodeBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NaturalCodeBench", "title": "NaturalCodeBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1117-naturalproofs", "domain": "数学", "name": "NaturalProofs", "released": "2021-06-07", "url": "https://hub.opencompass.org.cn/dataset-detail/NaturalProofs"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NaturalProofs", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NaturalProofs", "title": "NaturalProofs"}, {"benchmarks": [{"benchmark_id": "opencompass-1074-newsbench", "domain": "创作", "name": "NewsBench", "released": "2024-06-04", "url": "https://hub.opencompass.org.cn/dataset-detail/NewsBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NewsBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NewsBench", "title": "NewsBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1533-nutritionqa", "domain": "多模态", "name": "NutritionQA", "released": "2025-02-19", "url": "https://hub.opencompass.org.cn/dataset-detail/NutritionQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NutritionQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/NutritionQA", "title": "NutritionQA"}, {"benchmarks": [{"benchmark_id": "opencompass-525-ocnli", "domain": "推理", "name": "OCNLI", "released": "2020-10-12", "url": "https://hub.opencompass.org.cn/dataset-detail/OCNLI"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OCNLI", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OCNLI", "title": "OCNLI"}, {"benchmarks": [{"benchmark_id": "opencompass-557-ocrbench", "domain": "语言", "name": "OCRBench", "released": "2024-01-17", "url": "https://hub.opencompass.org.cn/dataset-detail/OCRBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OCRBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OCRBench", "title": "OCRBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1990-opt-bench", "domain": "智能体", "name": "OPT-BENCH", "released": "2025-06-12", "url": "https://hub.opencompass.org.cn/dataset-detail/OPT-BENCH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OPT-BENCH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OPT-BENCH", "title": "OPT-BENCH"}, {"benchmarks": [{"benchmark_id": "opencompass-2086-or-bench", "domain": "理解", "name": "OR-Bench", "released": "2024-05-31", "url": "https://hub.opencompass.org.cn/dataset-detail/OR-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OR-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OR-Bench", "title": "OR-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2005-oss-bench", "domain": "代码", "name": "OSS-Bench", "released": "2025-06-30", "url": "https://hub.opencompass.org.cn/dataset-detail/OSS-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OSS-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OSS-Bench", "title": "OSS-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1705-olymmath", "domain": "强推理", "name": "OlymMATH", "released": "2025-03-28", "url": "https://hub.opencompass.org.cn/dataset-detail/OlymMATH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OlymMATH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OlymMATH", "title": "OlymMATH"}, {"benchmarks": [{"benchmark_id": "opencompass-1070-olympiadbench", "domain": "学科", "name": "OlympiadBench", "released": "2024-06-06", "url": "https://hub.opencompass.org.cn/dataset-detail/OlympiadBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OlympiadBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OlympiadBench", "title": "OlympiadBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1329-olympicarena", "domain": "推理", "name": "OlympicArena", "released": "2024-06-18", "url": "https://hub.opencompass.org.cn/dataset-detail/OlympicArena"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OlympicArena", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OlympicArena", "title": "OlympicArena"}, {"benchmarks": [{"benchmark_id": "opencompass-1244-omni-math", "domain": "强推理", "name": "Omni-MATH", "released": "2024-10-10", "url": "https://hub.opencompass.org.cn/dataset-detail/Omni-MATH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Omni-MATH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Omni-MATH", "title": "Omni-MATH"}, {"benchmarks": [{"benchmark_id": "opencompass-1553-omnialign-v", "domain": "多模态", "name": "OmniAlign-V", "released": "2025-02-18", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniAlign-V"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniAlign-V", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniAlign-V", "title": "OmniAlign-V"}, {"benchmarks": [{"benchmark_id": "opencompass-1981-omnibench", "domain": "智能体", "name": "OmniBench", "released": "2025-06-10", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniBench", "title": "OmniBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1942-omnidocbench", "domain": "多模态", "name": "OmniDocBench", "released": "2024-12-02", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniDocBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniDocBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniDocBench", "title": "OmniDocBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1801-omnigirl", "domain": "多模态", "name": "OmniGIRL", "released": "2025-05-08", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniGIRL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniGIRL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniGIRL", "title": "OmniGIRL"}, {"benchmarks": [{"benchmark_id": "opencompass-2142-omnimmi", "domain": "多模态", "name": "OmniMMI", "released": "2025-04-01", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniMMI"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniMMI", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OmniMMI", "title": "OmniMMI"}, {"benchmarks": [{"benchmark_id": "opencompass-631-openfindata", "domain": "知识", "name": "OpenFinData", "released": "2023-12-29", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenFinData"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenFinData", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenFinData", "title": "OpenFinData"}, {"benchmarks": [{"benchmark_id": "opencompass-1754-openturingbench", "domain": "其他", "name": "OpenTuringBench", "released": "2025-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenTuringBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenTuringBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenTuringBench", "title": "OpenTuringBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1979-openunlearning", "domain": "其他", "name": "OpenUnlearning", "released": "2025-06-14", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenUnlearning"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenUnlearning", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenUnlearning", "title": "OpenUnlearning"}, {"benchmarks": [{"benchmark_id": "opencompass-518-openbookqa", "domain": "理解", "name": "OpenbookQA", "released": "2018-09-08", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenbookQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenbookQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/OpenbookQA", "title": "OpenbookQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1917-orak", "domain": "智能体", "name": "Orak", "released": "2025-06-04", "url": "https://hub.opencompass.org.cn/dataset-detail/Orak"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Orak", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Orak", "title": "Orak"}, {"benchmarks": [{"benchmark_id": "opencompass-1283-p-mmeval", "domain": "语言", "name": "P-MMEval", "released": "2024-11-14", "url": "https://hub.opencompass.org.cn/dataset-detail/P-MMEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/P-MMEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/P-MMEval", "title": "P-MMEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1076-pca-bench", "domain": "其他", "name": "PCA-Bench", "released": "2024-02-21", "url": "https://hub.opencompass.org.cn/dataset-detail/PCA-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PCA-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PCA-Bench", "title": "PCA-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2350-picabench", "domain": "多模态", "name": "PICABench", "released": "2025-10-27", "url": "https://hub.opencompass.org.cn/dataset-detail/PICABench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PICABench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PICABench", "title": "PICABench"}, {"benchmarks": [{"benchmark_id": "opencompass-532-piqa", "domain": "推理", "name": "PIQA", "released": "2019-11-26", "url": "https://hub.opencompass.org.cn/dataset-detail/PIQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PIQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PIQA", "title": "PIQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1362-pope", "domain": "多模态", "name": "POPE", "released": "2023-05-17", "url": "https://hub.opencompass.org.cn/dataset-detail/POPE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/POPE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/POPE", "title": "POPE"}, {"benchmarks": [{"benchmark_id": "opencompass-1681-prmbench-preview", "domain": "学科", "name": "PRMBench_Preview", "released": "2025-01-06", "url": "https://hub.opencompass.org.cn/dataset-detail/PRMBench_Preview"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PRMBench_Preview", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PRMBench_Preview", "title": "PRMBench_Preview"}, {"benchmarks": [{"benchmark_id": "opencompass-1730-paperbench", "domain": "理解", "name": "PaperBench", "released": "2025-04-07", "url": "https://hub.opencompass.org.cn/dataset-detail/PaperBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PaperBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PaperBench", "title": "PaperBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1836-pashtoocr", "domain": "多模态", "name": "PashtoOCR", "released": "2025-05-01", "url": "https://hub.opencompass.org.cn/dataset-detail/PashtoOCR"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PashtoOCR", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PashtoOCR", "title": "PashtoOCR"}, {"benchmarks": [{"benchmark_id": "opencompass-1959-personalens", "domain": "语言", "name": "PersonaLens", "released": "2025-06-11", "url": "https://hub.opencompass.org.cn/dataset-detail/PersonaLens"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PersonaLens", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PersonaLens", "title": "PersonaLens"}, {"benchmarks": [{"benchmark_id": "opencompass-2087-perteval-scfm", "domain": "其他", "name": "PertEval-scFM", "released": "2024-10-03", "url": "https://hub.opencompass.org.cn/dataset-detail/PertEval-scFM"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PertEval-scFM", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PertEval-scFM", "title": "PertEval-scFM"}, {"benchmarks": [{"benchmark_id": "opencompass-2080-phygenbench", "domain": "多模态", "name": "PhyGenBench", "released": "2024-10-07", "url": "https://hub.opencompass.org.cn/dataset-detail/PhyGenBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PhyGenBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PhyGenBench", "title": "PhyGenBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1576-physreason", "domain": "学科", "name": "PhysReason", "released": "2025-02-17", "url": "https://hub.opencompass.org.cn/dataset-detail/PhysReason"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PhysReason", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PhysReason", "title": "PhysReason"}, {"benchmarks": [{"benchmark_id": "opencompass-1251-planbench", "domain": "强推理", "name": "PlanBench", "released": "2022-06-21", "url": "https://hub.opencompass.org.cn/dataset-detail/PlanBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PlanBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PlanBench", "title": "PlanBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1675-pokerbench", "domain": "其他", "name": "PokerBench", "released": "2025-01-24", "url": "https://hub.opencompass.org.cn/dataset-detail/PokerBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PokerBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PokerBench", "title": "PokerBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1579-postersum", "domain": "多模态", "name": "PosterSum", "released": "2025-02-24", "url": "https://hub.opencompass.org.cn/dataset-detail/PosterSum"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PosterSum", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/PosterSum", "title": "PosterSum"}, {"benchmarks": [{"benchmark_id": "opencompass-1632-probench", "domain": "多模态", "name": "ProBench", "released": "2025-03-10", "url": "https://hub.opencompass.org.cn/dataset-detail/ProBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProBench", "title": "ProBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1633-projudge", "domain": "多模态", "name": "ProJudge", "released": "2025-03-09", "url": "https://hub.opencompass.org.cn/dataset-detail/ProJudge"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProJudge", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProJudge", "title": "ProJudge"}, {"benchmarks": [{"benchmark_id": "opencompass-1621-processbench", "domain": "学科", "name": "ProcessBench", "released": "2024-12-10", "url": "https://hub.opencompass.org.cn/dataset-detail/ProcessBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProcessBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProcessBench", "title": "ProcessBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1120-proofnet", "domain": "数学", "name": "ProofNet", "released": "2023-02-24", "url": "https://hub.opencompass.org.cn/dataset-detail/ProofNet"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProofNet", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ProofNet", "title": "ProofNet"}, {"benchmarks": [{"benchmark_id": "opencompass-1000-q-bench", "domain": "理解", "name": "Q-Bench", "released": "2024-08-16", "url": "https://hub.opencompass.org.cn/dataset-detail/Q-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Q-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Q-Bench", "title": "Q-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1103-qasc", "domain": "知识", "name": "QASC", "released": "2020-02-04", "url": "https://hub.opencompass.org.cn/dataset-detail/QASC"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/QASC", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/QASC", "title": "QASC"}, {"benchmarks": [{"benchmark_id": "opencompass-516-race-high", "domain": "理解", "name": "RACE(High)", "released": "2017-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28High%29"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RACE%28High%29", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28High%29", "title": "RACE(High)"}, {"benchmarks": [{"benchmark_id": "opencompass-517-race-middle", "domain": "理解", "name": "RACE(Middle)", "released": "2017-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28Middle%29"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RACE%28Middle%29", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28Middle%29", "title": "RACE(Middle)"}, {"benchmarks": [{"benchmark_id": "opencompass-1918-rdb2g-bench", "domain": "代码", "name": "RDB2G-Bench", "released": "2025-06-02", "url": "https://hub.opencompass.org.cn/dataset-detail/RDB2G-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RDB2G-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RDB2G-Bench", "title": "RDB2G-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1252-re-bench", "domain": "强推理", "name": "RE-Bench", "released": "2024-11-22", "url": "https://hub.opencompass.org.cn/dataset-detail/RE-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RE-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RE-Bench", "title": "RE-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1755-real", "domain": "智能体", "name": "REAL", "released": "2025-04-17", "url": "https://hub.opencompass.org.cn/dataset-detail/REAL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/REAL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/REAL", "title": "REAL"}, {"benchmarks": [{"benchmark_id": "opencompass-2033-rexbench", "domain": "代码", "name": "RExBench", "released": "2025-06-27", "url": "https://hub.opencompass.org.cn/dataset-detail/RExBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RExBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RExBench", "title": "RExBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1656-rfuav", "domain": "其他", "name": "RFUAV", "released": "2025-03-18", "url": "https://hub.opencompass.org.cn/dataset-detail/RFUAV"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RFUAV", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RFUAV", "title": "RFUAV"}, {"benchmarks": [{"benchmark_id": "opencompass-2048-risebench", "domain": "推理", "name": "RISEBench", "released": "2025-04-03", "url": "https://hub.opencompass.org.cn/dataset-detail/RISEBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RISEBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RISEBench", "title": "RISEBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1524-rm-bench", "domain": "语言", "name": "RM-Bench", "released": "2024-10-21", "url": "https://hub.opencompass.org.cn/dataset-detail/RM-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RM-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RM-Bench", "title": "RM-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1678-rsmmvp", "domain": "多模态", "name": "RSMMVP", "released": "2025-03-24", "url": "https://hub.opencompass.org.cn/dataset-detail/RSMMVP"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RSMMVP", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RSMMVP", "title": "RSMMVP"}, {"benchmarks": [{"benchmark_id": "opencompass-528-rte", "domain": "推理", "name": "RTE", "released": null, "url": "https://hub.opencompass.org.cn/dataset-detail/RTE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RTE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RTE", "title": "RTE"}, {"benchmarks": [{"benchmark_id": "opencompass-1738-rulistening", "domain": "理解", "name": "RUListening", "released": "2025-04-01", "url": "https://hub.opencompass.org.cn/dataset-detail/RUListening"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RUListening", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RUListening", "title": "RUListening"}, {"benchmarks": [{"benchmark_id": "opencompass-1707-rxrx3-core", "domain": "其他", "name": "RXRX3-CORE", "released": "2025-03-26", "url": "https://hub.opencompass.org.cn/dataset-detail/RXRX3-CORE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RXRX3-CORE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RXRX3-CORE", "title": "RXRX3-CORE"}, {"benchmarks": [{"benchmark_id": "opencompass-530-record", "domain": "推理", "name": "ReCoRD", "released": "2018-10-30", "url": "https://hub.opencompass.org.cn/dataset-detail/ReCoRD"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ReCoRD", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ReCoRD", "title": "ReCoRD"}, {"benchmarks": [{"benchmark_id": "opencompass-1124-realtoxicityprompts", "domain": "安全", "name": "RealToxicityPrompts", "released": "2020-11-16", "url": "https://hub.opencompass.org.cn/dataset-detail/RealToxicityPrompts"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RealToxicityPrompts", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RealToxicityPrompts", "title": "RealToxicityPrompts"}, {"benchmarks": [{"benchmark_id": "opencompass-1361-realworldqa", "domain": "多模态", "name": "RealworldQA", "released": "2024-04-12", "url": "https://hub.opencompass.org.cn/dataset-detail/RealworldQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RealworldQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RealworldQA", "title": "RealworldQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1333-redcode", "domain": "安全", "name": "RedCode", "released": "2024-11-12", "url": "https://hub.opencompass.org.cn/dataset-detail/RedCode"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RedCode", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RedCode", "title": "RedCode"}, {"benchmarks": [{"benchmark_id": "opencompass-1317-repliqa", "domain": "推理", "name": "RepLiQA", "released": "2024-06-17", "url": "https://hub.opencompass.org.cn/dataset-detail/RepLiQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RepLiQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RepLiQA", "title": "RepLiQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1087-reveal", "domain": "推理", "name": "Reveal", "released": "2024-05-21", "url": "https://hub.opencompass.org.cn/dataset-detail/Reveal"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Reveal", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Reveal", "title": "Reveal"}, {"benchmarks": [{"benchmark_id": "opencompass-1915-rewardbench", "domain": "指令跟随", "name": "RewardBench", "released": "2025-06-02", "url": "https://hub.opencompass.org.cn/dataset-detail/RewardBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RewardBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RewardBench", "title": "RewardBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2371-rigorousbench", "domain": "智能体", "name": "RigorousBench", "released": "2025-10-02", "url": "https://hub.opencompass.org.cn/dataset-detail/RigorousBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RigorousBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RigorousBench", "title": "RigorousBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1091-rolellm", "domain": "其他", "name": "RoleLLM", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/RoleLLM"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RoleLLM", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/RoleLLM", "title": "RoleLLM"}, {"benchmarks": [{"benchmark_id": "opencompass-1003-s-eval", "domain": "安全", "name": "S-Eval", "released": "2024-05-23", "url": "https://hub.opencompass.org.cn/dataset-detail/S-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/S-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/S-Eval", "title": "S-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-1772-s1-bench", "domain": "强推理", "name": "S1-Bench", "released": "2025-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/S1-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/S1-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/S1-Bench", "title": "S1-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2083-saebench", "domain": "理解", "name": "SAEBench", "released": "2025-03-12", "url": "https://hub.opencompass.org.cn/dataset-detail/SAEBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SAEBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SAEBench", "title": "SAEBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1077-salad-bench", "domain": "安全", "name": "SALAD-Bench", "released": "2024-02-07", "url": "https://hub.opencompass.org.cn/dataset-detail/SALAD-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SALAD-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SALAD-Bench", "title": "SALAD-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1732-scam", "domain": "多模态", "name": "SCAM", "released": "2025-04-07", "url": "https://hub.opencompass.org.cn/dataset-detail/SCAM"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SCAM", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SCAM", "title": "SCAM"}, {"benchmarks": [{"benchmark_id": "opencompass-1986-sec-bench", "domain": "安全", "name": "SEC-bench", "released": "2025-06-13", "url": "https://hub.opencompass.org.cn/dataset-detail/SEC-bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEC-bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEC-bench", "title": "SEC-bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1359-seed-bench", "domain": "多模态", "name": "SEED-Bench", "released": "2023-07-30", "url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEED-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench", "title": "SEED-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1363-seed-bench-2", "domain": "多模态", "name": "SEED-Bench-2", "released": "2023-11-28", "url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2", "title": "SEED-Bench-2"}, {"benchmarks": [{"benchmark_id": "opencompass-1395-seed-bench-2-plus", "domain": "多模态", "name": "SEED-Bench-2-Plus", "released": "2024-04-25", "url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2-Plus"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2-Plus", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2-Plus", "title": "SEED-Bench-2-Plus"}, {"benchmarks": [{"benchmark_id": "opencompass-1961-sfe", "domain": "多模态", "name": "SFE", "released": "2025-06-12", "url": "https://hub.opencompass.org.cn/dataset-detail/SFE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SFE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SFE", "title": "SFE"}, {"benchmarks": [{"benchmark_id": "opencompass-1330-sg-bench", "domain": "安全", "name": "SG-Bench", "released": "2024-10-29", "url": "https://hub.opencompass.org.cn/dataset-detail/SG-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SG-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SG-Bench", "title": "SG-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2325-sgi-bench", "domain": "多模态", "name": "SGI-Bench", "released": "2025-12-22", "url": "https://hub.opencompass.org.cn/dataset-detail/SGI-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SGI-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SGI-Bench", "title": "SGI-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-533-siqa", "domain": "推理", "name": "SIQA", "released": "2019-09-09", "url": "https://hub.opencompass.org.cn/dataset-detail/SIQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SIQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SIQA", "title": "SIQA"}, {"benchmarks": [{"benchmark_id": "opencompass-2029-smmile", "domain": "多模态", "name": "SMMILE", "released": "2025-06-26", "url": "https://hub.opencompass.org.cn/dataset-detail/SMMILE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SMMILE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SMMILE", "title": "SMMILE"}, {"benchmarks": [{"benchmark_id": "opencompass-1845-stark-10k", "domain": "多模态", "name": "STARK_10k", "released": "2025-05-16", "url": "https://hub.opencompass.org.cn/dataset-detail/STARK_10k"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/STARK_10k", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/STARK_10k", "title": "STARK_10k"}, {"benchmarks": [{"benchmark_id": "opencompass-1113-svamp", "domain": "数学", "name": "SVAMP", "released": "2021-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/SVAMP"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SVAMP", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SVAMP", "title": "SVAMP"}, {"benchmarks": [{"benchmark_id": "opencompass-1989-swe-factory", "domain": "代码", "name": "SWE-Factory", "released": "2025-06-12", "url": "https://hub.opencompass.org.cn/dataset-detail/SWE-Factory"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SWE-Factory", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SWE-Factory", "title": "SWE-Factory"}, {"benchmarks": [{"benchmark_id": "opencompass-1984-swe-bench-live", "domain": "代码", "name": "SWE-bench-Live", "released": "2025-06-01", "url": "https://hub.opencompass.org.cn/dataset-detail/SWE-bench-Live"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SWE-bench-Live", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SWE-bench-Live", "title": "SWE-bench-Live"}, {"benchmarks": [{"benchmark_id": "opencompass-1073-safetybench", "domain": "安全", "name": "SafetyBench", "released": "2024-06-24", "url": "https://hub.opencompass.org.cn/dataset-detail/SafetyBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SafetyBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SafetyBench", "title": "SafetyBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1323-scifibench", "domain": "多模态", "name": "SciFIBench", "released": "2024-05-14", "url": "https://hub.opencompass.org.cn/dataset-detail/SciFIBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SciFIBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SciFIBench", "title": "SciFIBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1100-scienceqa", "domain": "知识", "name": "ScienceQA", "released": "2022-10-17", "url": "https://hub.opencompass.org.cn/dataset-detail/ScienceQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ScienceQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ScienceQA", "title": "ScienceQA"}, {"benchmarks": [{"benchmark_id": "opencompass-948-secbench", "domain": "安全", "name": "SecBench", "released": "2024-01-19", "url": "https://hub.opencompass.org.cn/dataset-detail/SecBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SecBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SecBench", "title": "SecBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2348-shell", "domain": "安全", "name": "Shell", "released": "2025-12-25", "url": "https://hub.opencompass.org.cn/dataset-detail/Shell"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Shell", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Shell", "title": "Shell"}, {"benchmarks": [{"benchmark_id": "opencompass-1321-shoppingmmlu", "domain": "其他", "name": "ShoppingMMLU", "released": "2024-10-28", "url": "https://hub.opencompass.org.cn/dataset-detail/ShoppingMMLU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ShoppingMMLU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ShoppingMMLU", "title": "ShoppingMMLU"}, {"benchmarks": [{"benchmark_id": "opencompass-2217-smartbench", "domain": "多模态", "name": "SmartBench", "released": "2025-09-26", "url": "https://hub.opencompass.org.cn/dataset-detail/SmartBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SmartBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SmartBench", "title": "SmartBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2092-spatial457", "domain": "多模态", "name": "Spatial457", "released": "2025-04-09", "url": "https://hub.opencompass.org.cn/dataset-detail/Spatial457"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Spatial457", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Spatial457", "title": "Spatial457"}, {"benchmarks": [{"benchmark_id": "opencompass-1133-spider", "domain": "智能体", "name": "Spider", "released": "2019-02-02", "url": "https://hub.opencompass.org.cn/dataset-detail/Spider"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Spider", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Spider", "title": "Spider"}, {"benchmarks": [{"benchmark_id": "opencompass-1270-spider2-v", "domain": "智能体", "name": "Spider2-V", "released": "2024-07-15", "url": "https://hub.opencompass.org.cn/dataset-detail/Spider2-V"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Spider2-V", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Spider2-V", "title": "Spider2-V"}, {"benchmarks": [{"benchmark_id": "opencompass-1152-sportqa", "domain": "理解", "name": "SportQA", "released": "2024-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/SportQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SportQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SportQA", "title": "SportQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1267-spreadsheetbench", "domain": "其他", "name": "SpreadsheetBench", "released": "2024-06-21", "url": "https://hub.opencompass.org.cn/dataset-detail/SpreadsheetBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SpreadsheetBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SpreadsheetBench", "title": "SpreadsheetBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1084-stabletoolbench", "domain": "其他", "name": "StableToolBench", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/StableToolBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StableToolBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StableToolBench", "title": "StableToolBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1107-strategyqa", "domain": "知识", "name": "StrategyQA", "released": "2021-01-06", "url": "https://hub.opencompass.org.cn/dataset-detail/StrategyQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StrategyQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StrategyQA", "title": "StrategyQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1542-structflowbench", "domain": "指令跟随", "name": "StructFlowBench", "released": "2025-02-20", "url": "https://hub.opencompass.org.cn/dataset-detail/StructFlowBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StructFlowBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StructFlowBench", "title": "StructFlowBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2082-structtokenbench", "domain": "其他", "name": "StructTokenBench", "released": "2025-02-28", "url": "https://hub.opencompass.org.cn/dataset-detail/StructTokenBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StructTokenBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StructTokenBench", "title": "StructTokenBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1082-studenteval", "domain": "代码", "name": "StudentEval", "released": "2024-08-11", "url": "https://hub.opencompass.org.cn/dataset-detail/StudentEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StudentEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StudentEval", "title": "StudentEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1744-stylerec", "domain": "理解", "name": "StyleRec", "released": "2025-04-15", "url": "https://hub.opencompass.org.cn/dataset-detail/StyleRec"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StyleRec", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/StyleRec", "title": "StyleRec"}, {"benchmarks": [{"benchmark_id": "opencompass-1538-supergpqa", "domain": "学科", "name": "SuperGPQA", "released": "2025-02-20", "url": "https://hub.opencompass.org.cn/dataset-detail/SuperGPQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SuperGPQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SuperGPQA", "title": "SuperGPQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1608-swiltra-bench", "domain": "语言", "name": "SwiLTra-Bench", "released": "2025-03-03", "url": "https://hub.opencompass.org.cn/dataset-detail/SwiLTra-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SwiLTra-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/SwiLTra-Bench", "title": "SwiLTra-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-540-t-eval", "domain": "智能体", "name": "T-Eval", "released": "2024-01-15", "url": "https://hub.opencompass.org.cn/dataset-detail/T-Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/T-Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/T-Eval", "title": "T-Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-1846-tglg", "domain": "多模态", "name": "TGLG", "released": "2025-05-09", "url": "https://hub.opencompass.org.cn/dataset-detail/TGLG"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TGLG", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TGLG", "title": "TGLG"}, {"benchmarks": [{"benchmark_id": "opencompass-2050-thunder", "domain": "理解", "name": "THUNDER", "released": "2025-07-10", "url": "https://hub.opencompass.org.cn/dataset-detail/THUNDER"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/THUNDER", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/THUNDER", "title": "THUNDER"}, {"benchmarks": [{"benchmark_id": "opencompass-1132-tabfact", "domain": "智能体", "name": "TabFact", "released": "2020-06-14", "url": "https://hub.opencompass.org.cn/dataset-detail/TabFact"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TabFact", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TabFact", "title": "TabFact"}, {"benchmarks": [{"benchmark_id": "opencompass-2025-tableeval", "domain": "推理", "name": "TableEval", "released": "2025-06-05", "url": "https://hub.opencompass.org.cn/dataset-detail/TableEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TableEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TableEval", "title": "TableEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1164-taskbench", "domain": "智能体", "name": "TaskBench", "released": "2023-11-30", "url": "https://hub.opencompass.org.cn/dataset-detail/TaskBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TaskBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TaskBench", "title": "TaskBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1574-text2world", "domain": "其他", "name": "Text2World", "released": "2025-02-18", "url": "https://hub.opencompass.org.cn/dataset-detail/Text2World"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Text2World", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Text2World", "title": "Text2World"}, {"benchmarks": [{"benchmark_id": "opencompass-1734-thai-local-benchmark", "domain": "语言", "name": "Thai_local_benchmark", "released": "2025-04-08", "url": "https://hub.opencompass.org.cn/dataset-detail/Thai_local_benchmark"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Thai_local_benchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Thai_local_benchmark", "title": "Thai_local_benchmark"}, {"benchmarks": [{"benchmark_id": "opencompass-1134-theoremqa", "domain": "推理", "name": "TheoremQA", "released": "2023-12-06", "url": "https://hub.opencompass.org.cn/dataset-detail/TheoremQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TheoremQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TheoremQA", "title": "TheoremQA"}, {"benchmarks": [{"benchmark_id": "opencompass-2391-threat-signature-eval", "domain": "视觉定位", "name": "Threat-Signature_Eval", "released": "2025-11-19", "url": "https://hub.opencompass.org.cn/dataset-detail/Threat-Signature_Eval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Threat-Signature_Eval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Threat-Signature_Eval", "title": "Threat-Signature_Eval"}, {"benchmarks": [{"benchmark_id": "opencompass-1668-timetravel", "domain": "多模态", "name": "TimeTravel", "released": "2025-02-20", "url": "https://hub.opencompass.org.cn/dataset-detail/TimeTravel"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TimeTravel", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TimeTravel", "title": "TimeTravel"}, {"benchmarks": [{"benchmark_id": "opencompass-1712-toolhop", "domain": "智能体", "name": "ToolHop", "released": "2025-01-05", "url": "https://hub.opencompass.org.cn/dataset-detail/ToolHop"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ToolHop", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ToolHop", "title": "ToolHop"}, {"benchmarks": [{"benchmark_id": "opencompass-1607-toolret", "domain": "其他", "name": "ToolRet", "released": "2025-03-03", "url": "https://hub.opencompass.org.cn/dataset-detail/ToolRet"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ToolRet", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ToolRet", "title": "ToolRet"}, {"benchmarks": [{"benchmark_id": "opencompass-1855-transbench", "domain": "语言", "name": "TransBench", "released": "2025-05-20", "url": "https://hub.opencompass.org.cn/dataset-detail/TransBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TransBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TransBench", "title": "TransBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2039-translaw", "domain": "语言", "name": "TransLaw", "released": "2025-07-01", "url": "https://hub.opencompass.org.cn/dataset-detail/TransLaw"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TransLaw", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TransLaw", "title": "TransLaw"}, {"benchmarks": [{"benchmark_id": "opencompass-512-triviaqa", "domain": "知识", "name": "TriviaQA", "released": "2017-05-09", "url": "https://hub.opencompass.org.cn/dataset-detail/TriviaQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TriviaQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TriviaQA", "title": "TriviaQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1096-truthfulqa", "domain": "知识", "name": "TruthfulQA", "released": "2022-05-08", "url": "https://hub.opencompass.org.cn/dataset-detail/TruthfulQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TruthfulQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TruthfulQA", "title": "TruthfulQA"}, {"benchmarks": [{"benchmark_id": "opencompass-508-tydiqa", "domain": "知识", "name": "TyDiQA", "released": "2020-03-10", "url": "https://hub.opencompass.org.cn/dataset-detail/TyDiQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TyDiQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/TyDiQA", "title": "TyDiQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1742-u-niah", "domain": "长文本", "name": "U-NIAH", "released": "2025-03-01", "url": "https://hub.opencompass.org.cn/dataset-detail/U-NIAH"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/U-NIAH", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/U-NIAH", "title": "U-NIAH"}, {"benchmarks": [{"benchmark_id": "opencompass-1088-uhgeval", "domain": "知识", "name": "UHGEval", "released": "2024-05-24", "url": "https://hub.opencompass.org.cn/dataset-detail/UHGEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UHGEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UHGEval", "title": "UHGEval"}, {"benchmarks": [{"benchmark_id": "opencompass-2014-utboost", "domain": "代码", "name": "UTBoost", "released": "2025-06-10", "url": "https://hub.opencompass.org.cn/dataset-detail/UTBoost"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UTBoost", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UTBoost", "title": "UTBoost"}, {"benchmarks": [{"benchmark_id": "opencompass-1332-unibench", "domain": "多模态", "name": "UniBench", "released": "2024-08-09", "url": "https://hub.opencompass.org.cn/dataset-detail/UniBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UniBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UniBench", "title": "UniBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1636-urbanvideo-bench", "domain": "多模态", "name": "UrbanVideo-Bench", "released": "2025-03-08", "url": "https://hub.opencompass.org.cn/dataset-detail/UrbanVideo-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UrbanVideo-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/UrbanVideo-Bench", "title": "UrbanVideo-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1654-v-star", "domain": "多模态", "name": "V-STaR", "released": "2025-03-14", "url": "https://hub.opencompass.org.cn/dataset-detail/V-STaR"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/V-STaR", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/V-STaR", "title": "V-STaR"}, {"benchmarks": [{"benchmark_id": "opencompass-1375-vbench", "domain": "多模态", "name": "VBench", "released": "2023-11-29", "url": "https://hub.opencompass.org.cn/dataset-detail/VBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VBench", "title": "VBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1875-vibe", "domain": "其他", "name": "VIBE", "released": "2025-05-29", "url": "https://hub.opencompass.org.cn/dataset-detail/VIBE"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VIBE", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VIBE", "title": "VIBE"}, {"benchmarks": [{"benchmark_id": "opencompass-2063-visco", "domain": "多模态", "name": "VISCO", "released": "2024-12-03", "url": "https://hub.opencompass.org.cn/dataset-detail/VISCO"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VISCO", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VISCO", "title": "VISCO"}, {"benchmarks": [{"benchmark_id": "opencompass-2388-vknowu", "domain": "多模态", "name": "VKnowU", "released": "2025-11-25", "url": "https://hub.opencompass.org.cn/dataset-detail/VKnowU"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VKnowU", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VKnowU", "title": "VKnowU"}, {"benchmarks": [{"benchmark_id": "opencompass-1541-vlm2-bench", "domain": "多模态", "name": "VLM2-Bench", "released": "2025-02-17", "url": "https://hub.opencompass.org.cn/dataset-detail/VLM2-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VLM2-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VLM2-Bench", "title": "VLM2-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-2383-vrbench", "domain": "多模态", "name": "VRBench", "released": "2025-06-12", "url": "https://hub.opencompass.org.cn/dataset-detail/VRBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VRBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VRBench", "title": "VRBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1717-vilbench", "domain": "多模态", "name": "ViLBench", "released": "2025-03-26", "url": "https://hub.opencompass.org.cn/dataset-detail/ViLBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ViLBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ViLBench", "title": "ViLBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1993-vistorybench", "domain": "多模态", "name": "ViStoryBench", "released": "2025-06-25", "url": "https://hub.opencompass.org.cn/dataset-detail/ViStoryBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ViStoryBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ViStoryBench", "title": "ViStoryBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1358-video-mme", "domain": "多模态", "name": "Video-MME", "released": "2024-03-31", "url": "https://hub.opencompass.org.cn/dataset-detail/Video-MME"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Video-MME", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Video-MME", "title": "Video-MME"}, {"benchmarks": [{"benchmark_id": "opencompass-1277-videogui", "domain": "创作", "name": "VideoGUI", "released": "2024-06-14", "url": "https://hub.opencompass.org.cn/dataset-detail/VideoGUI"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VideoGUI", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VideoGUI", "title": "VideoGUI"}, {"benchmarks": [{"benchmark_id": "opencompass-1926-videomathqa", "domain": "多模态", "name": "VideoMathQA", "released": "2025-06-05", "url": "https://hub.opencompass.org.cn/dataset-detail/VideoMathQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VideoMathQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VideoMathQA", "title": "VideoMathQA"}, {"benchmarks": [{"benchmark_id": "opencompass-1900-videoreasonbench", "domain": "强推理", "name": "VideoReasonBench", "released": "2025-06-05", "url": "https://hub.opencompass.org.cn/dataset-detail/VideoReasonBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VideoReasonBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VideoReasonBench", "title": "VideoReasonBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1748-visualpuzzles", "domain": "强推理", "name": "VisualPuzzles", "released": "2025-04-16", "url": "https://hub.opencompass.org.cn/dataset-detail/VisualPuzzles"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VisualPuzzles", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VisualPuzzles", "title": "VisualPuzzles"}, {"benchmarks": [{"benchmark_id": "opencompass-1634-visualsimpleqa", "domain": "多模态", "name": "VisualSimpleQA", "released": "2025-03-09", "url": "https://hub.opencompass.org.cn/dataset-detail/VisualSimpleQA"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VisualSimpleQA", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/VisualSimpleQA", "title": "VisualSimpleQA"}, {"benchmarks": [{"benchmark_id": "opencompass-507-wsc", "domain": "语言", "name": "WSC", "released": null, "url": "https://hub.opencompass.org.cn/dataset-detail/WSC"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WSC", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WSC", "title": "WSC"}, {"benchmarks": [{"benchmark_id": "opencompass-1982-webui-bench", "domain": "代码", "name": "WebUI-Bench", "released": "2025-06-09", "url": "https://hub.opencompass.org.cn/dataset-detail/WebUI-Bench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WebUI-Bench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WebUI-Bench", "title": "WebUI-Bench"}, {"benchmarks": [{"benchmark_id": "opencompass-1279-whodunitbench", "domain": "智能体", "name": "WhodunitBench", "released": "2024-09-26", "url": "https://hub.opencompass.org.cn/dataset-detail/WhodunitBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WhodunitBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WhodunitBench", "title": "WhodunitBench"}, {"benchmarks": [{"benchmark_id": "opencompass-504-wic", "domain": "语言", "name": "WiC", "released": "2018-08-28", "url": "https://hub.opencompass.org.cn/dataset-detail/WiC"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WiC", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WiC", "title": "WiC"}, {"benchmarks": [{"benchmark_id": "opencompass-1318-wikicontradict", "domain": "推理", "name": "WikiContradict", "released": "2024-06-19", "url": "https://hub.opencompass.org.cn/dataset-detail/WikiContradict"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WikiContradict", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WikiContradict", "title": "WikiContradict"}, {"benchmarks": [{"benchmark_id": "opencompass-1129-wikisql", "domain": "其他", "name": "WikiSQL", "released": "2017-11-09", "url": "https://hub.opencompass.org.cn/dataset-detail/WikiSQL"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WikiSQL", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WikiSQL", "title": "WikiSQL"}, {"benchmarks": [{"benchmark_id": "opencompass-1555-wildbench", "domain": "理解", "name": "WildBench", "released": "2024-06-07", "url": "https://hub.opencompass.org.cn/dataset-detail/WildBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WildBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WildBench", "title": "WildBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2445-wildclawbench", "domain": "智能体", "name": "WildClawBench", "released": "2026-04-07", "url": "https://hub.opencompass.org.cn/dataset-detail/WildClawBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WildClawBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WildClawBench", "title": "WildClawBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1109-winogrande", "domain": "推理", "name": "WinoGrande", "released": "2019-11-21", "url": "https://hub.opencompass.org.cn/dataset-detail/WinoGrande"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WinoGrande", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WinoGrande", "title": "WinoGrande"}, {"benchmarks": [{"benchmark_id": "opencompass-1938-worldgenbench", "domain": "多模态", "name": "WorldGenBench", "released": "2025-06-16", "url": "https://hub.opencompass.org.cn/dataset-detail/WorldGenBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WorldGenBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WorldGenBench", "title": "WorldGenBench"}, {"benchmarks": [{"benchmark_id": "opencompass-1735-worldscore", "domain": "创作", "name": "WorldScore", "released": "2025-04-01", "url": "https://hub.opencompass.org.cn/dataset-detail/WorldScore"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WorldScore", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WorldScore", "title": "WorldScore"}, {"benchmarks": [{"benchmark_id": "opencompass-1701-writingbench", "domain": "长文本", "name": "WritingBench", "released": "2025-03-07", "url": "https://hub.opencompass.org.cn/dataset-detail/WritingBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WritingBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/WritingBench", "title": "WritingBench"}, {"benchmarks": [{"benchmark_id": "opencompass-521-xsum", "domain": "理解", "name": "XSum", "released": "2018-08-27", "url": "https://hub.opencompass.org.cn/dataset-detail/XSum"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/XSum", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/XSum", "title": "XSum"}, {"benchmarks": [{"benchmark_id": "opencompass-1017-yue-benchmark", "domain": "语言", "name": "Yue_Benchmark", "released": "2024-08-31", "url": "https://hub.opencompass.org.cn/dataset-detail/Yue_Benchmark"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Yue_Benchmark", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/Yue_Benchmark", "title": "Yue_Benchmark"}, {"benchmarks": [{"benchmark_id": "opencompass-1532-zerobench", "domain": "多模态", "name": "ZeroBench", "released": "2025-02-13", "url": "https://hub.opencompass.org.cn/dataset-detail/ZeroBench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ZeroBench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ZeroBench", "title": "ZeroBench"}, {"benchmarks": [{"benchmark_id": "opencompass-2127-clembench", "domain": "创作", "name": "clembench", "released": "2025-07-11", "url": "https://hub.opencompass.org.cn/dataset-detail/clembench"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/clembench", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/clembench", "title": "clembench"}, {"benchmarks": [{"benchmark_id": "opencompass-2128-lm-evaluation-harness", "domain": "知识", "name": "lm-evaluation-harness", "released": "2025-07-11", "url": "https://hub.opencompass.org.cn/dataset-detail/lm-evaluation-harness"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/lm-evaluation-harness", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/lm-evaluation-harness", "title": "lm-evaluation-harness"}, {"benchmarks": [{"benchmark_id": "opencompass-1518-minictx", "domain": "强推理", "name": "miniCTX", "released": "2024-08-03", "url": "https://hub.opencompass.org.cn/dataset-detail/miniCTX"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/miniCTX", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/miniCTX", "title": "miniCTX"}, {"benchmarks": [{"benchmark_id": "opencompass-1839-tiny-qa-benchmark-pp", "domain": "推理", "name": "tiny_qa_benchmark_pp", "released": "2025-05-17", "url": "https://hub.opencompass.org.cn/dataset-detail/tiny_qa_benchmark_pp"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/tiny_qa_benchmark_pp", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/tiny_qa_benchmark_pp", "title": "tiny_qa_benchmark_pp"}, {"benchmarks": [{"benchmark_id": "opencompass-1625-ubuntu-osworld", "domain": "多模态", "name": "ubuntu_osworld", "released": "2024-04-11", "url": "https://hub.opencompass.org.cn/dataset-detail/ubuntu_osworld"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ubuntu_osworld", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/ubuntu_osworld", "title": "ubuntu_osworld"}, {"benchmarks": [{"benchmark_id": "opencompass-1072-xcodeeval", "domain": "代码", "name": "xCodeEval", "released": "2023-11-06", "url": "https://hub.opencompass.org.cn/dataset-detail/xCodeEval"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/xCodeEval", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/xCodeEval", "title": "xCodeEval"}, {"benchmarks": [{"benchmark_id": "opencompass-1784-xverify", "domain": "推理", "name": "xVerify", "released": "2025-04-14", "url": "https://hub.opencompass.org.cn/dataset-detail/xVerify"}], "document_type": "registry_page", "id": "opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/xVerify", "source": "opencompass_hub", "source_url": "https://hub.opencompass.org.cn/dataset-detail/xVerify", "title": "xVerify"}], "domains": {"3d": 1, "agent": 10, "agentic": 8, "agents": 55, "ai_research": 2, "audio": 2, "biology": 1, "code": 15, "coding": 9, "coding_agent": 13, "computer_use": 1, "creativity": 1, "factuality": 2, "finance": 2, "frontend_development": 1, "general": 16, "health": 2, "healthcare": 6, "human_preference": 2, "image-generation": 2, "image_to_text": 4, "instruction-following": 1, "instruction_following": 5, "intelligence-index": 11, "knowledge": 7, "language": 11, "legal": 22, "long_context": 60, "math": 83, "medical": 1, "memory": 3, "multilingual": 1, "multimodal": 163, "other": 4, "physics": 7, "productivity": 6, "professional": 4, "psychology": 1, "question_answering": 1, "reasoning": 213, "research": 1, "robotics": 1, "safety": 18, "science": 6, "scientific_agent": 1, "security": 4, "spatial_reasoning": 8, "speech_to_text": 4, "structured_output": 5, "summarization": 2, "tool_use": 7, "video": 1, "vision": 6, "代码": 29, "其他": 45, "创作": 7, "医学": 1, "多模态": 132, "学科": 19, "安全": 21, "强推理": 20, "指令跟随": 5, "推理": 49, "数学": 11, "智能体": 30, "理解": 32, "知识": 28, "科学": 1, "视觉定位": 1, "语言": 19, "长文本": 11}, "entries": [{"aliases": ["GPQA", "GPQA-Diamond", "GPQA Diamond"], "benchmark_id": "gpqa_diamond", "caveat": "198 questions in the Diamond split, so run-to-run variance is wide and a single reported number hides it. Approaching saturation at the 2026 frontier, where reported scores cluster in the high 80s and 90s.", "document_count": 27, "document_ids": ["model_reports:anthropic_claude_3_7_sonnet", "model_reports:anthropic_claude_4_system_card", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_2_5_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:google_gemma_4_model_card", "model_reports:meta_llama_4", "model_reports:mistral_large_3", "model_reports:mistral_medium_3", "model_reports:moonshot_kimi_k3_model_card", "model_reports:openai_gpt_4_1", "model_reports:openai_gpt_5_6_release", "model_reports:openai_gpt_5_system_card", "model_reports:openai_o3_o4_mini_system_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card", "model_reports:qwen3_technical_report", "model_reports:tencent_hy4_preview", "model_reports:xai_grok_4_5", "model_reports:xai_grok_4_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_model_card", "model_reports:zai_glm_5_paper"], "document_share": 0.022332506203473945, "domain": "science", "name": "GPQA Diamond", "organization_count": 11, "organizations": ["Anthropic", "DeepSeek", "Google", "Meta", "Mistral", "Moonshot AI", "OpenAI", "Qwen", "Tencent", "Z.ai", "xAI"], "rank": 1, "released": "2023-11-20", "source": "model_reports", "url": "https://arxiv.org/abs/2311.12022"}, {"aliases": ["HLE", "Humanity's Last Exam", "HLE-Full", "HLE-Verified"], "benchmark_id": "hle", "caveat": "Tool access and test-time compute budget move this score more than model capability does, so two reported figures are rarely comparable. Cards increasingly report no-tools and with-tools figures as separate numbers.", "document_count": 20, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_pro_0813_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_2_5_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:google_gemma_4_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:openai_gpt_5_6_release", "model_reports:openai_gpt_5_system_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card", "model_reports:tencent_hy4_preview", "model_reports:xai_grok_4_5", "model_reports:xai_grok_4_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_3_flash_model_card", "model_reports:zai_glm_5_model_card", "model_reports:zai_glm_5_paper"], "document_share": 0.016542597187758478, "domain": "reasoning", "name": "Humanity's Last Exam", "organization_count": 9, "organizations": ["Anthropic", "DeepSeek", "Google", "Moonshot AI", "OpenAI", "Qwen", "Tencent", "Z.ai", "xAI"], "rank": 2, "released": "2025-01-23", "source": "model_reports", "url": "https://lastexam.ai/"}, {"aliases": ["Terminal-Bench", "Terminal Bench", "TerminalBench", "Terminal-Bench 2.0", "Terminal-Bench 2.1", "Terminal Bench 2", "TerminalBench Hard", "Terminal-Bench 3.0", "Terminal Bench 3", "Frontier-Bench", "FrontierBench", "Frontier-Bench v0.1", "FrontierSWE"], "benchmark_id": "terminal_bench", "caveat": "Environment image, timeout and permitted commands change results independently of the model. Version and harness both matter: 2.0 and 2.1 are different instruments, and cards report Terminus, Claude Code and Codex harness numbers for the same version. Frontier-Bench is this series' next version rather than a separate benchmark: frontierbench.ai now bills it as \"Terminal-Bench 3.0 (formerly Frontier-Bench)\", built by the makers of Harbor and Terminal-Bench, so its mentions are counted here. Merging it did not raise this count: all three cards that named Frontier-Bench (Claude Opus 5, Claude Fable 5 and Mythos 5, Kimi K3) also report Terminal-Bench in the same document, and the counting unit is the document, so each still adds one. Being a new version, its task set is not comparable to a 2.x number. Not to be confused with Cognition's FrontierCode, a separate instrument recorded separately despite the shared prefix.", "document_count": 19, "document_ids": ["model_reports:anthropic_claude_4_system_card", "model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:anthropic_claude_opus_5_system_card", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_pro_0813_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:openai_gpt_5_6_release", "model_reports:openai_gpt_5_6_system_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card", "model_reports:tencent_hy4_preview", "model_reports:xai_grok_4_5", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_3_flash_model_card", "model_reports:zai_glm_5_model_card", "model_reports:zai_glm_5_paper"], "document_share": 0.015715467328370553, "domain": "agent", "name": "Terminal-Bench", "organization_count": 9, "organizations": ["Anthropic", "DeepSeek", "Google", "Moonshot AI", "OpenAI", "Qwen", "Tencent", "Z.ai", "xAI"], "rank": 3, "released": "2025-05-19", "source": "model_reports", "url": "https://www.tbench.ai/"}, {"aliases": ["SWE-bench Verified", "SWE-bench-Verified", "SWE Bench Verified", "SWE Verified"], "benchmark_id": "swe_bench_verified", "caveat": "Scaffold, tool permissions and time limit are part of the result; vendors run their own harnesses.", "document_count": 18, "document_ids": ["model_reports:anthropic_claude_3_7_sonnet", "model_reports:anthropic_claude_4_system_card", "model_reports:anthropic_claude_opus_5_system_card", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_2_5_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:openai_gpt_4_1", "model_reports:openai_gpt_5_6_release", "model_reports:openai_gpt_5_system_card", "model_reports:openai_o3_o4_mini_system_card", "model_reports:qwen3_5_model_card", "model_reports:xai_grok_4_model_card", "model_reports:zai_glm_5_model_card", "model_reports:zai_glm_5_paper"], "document_share": 0.01488833746898263, "domain": "coding_agent", "name": "SWE-bench Verified", "organization_count": 8, "organizations": ["Anthropic", "DeepSeek", "Google", "Moonshot AI", "OpenAI", "Qwen", "Z.ai", "xAI"], "rank": 4, "released": "2024-08-13", "source": "model_reports", "url": "https://openai.com/index/introducing-swe-bench-verified/"}, {"aliases": ["AIME", "AIME 2024", "AIME 2025", "AIME 2026"], "benchmark_id": "aime", "caveat": "pass@k, majority vote and Python tool access each shift this by double digits.", "document_count": 17, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:anthropic_claude_3_7_sonnet", "model_reports:anthropic_claude_4_system_card", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:google_gemini_2_5_report", "model_reports:google_gemma_4_model_card", "model_reports:openai_gpt_4_1", "model_reports:openai_gpt_5_system_card", "model_reports:openai_o3_o4_mini_system_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_technical_report", "model_reports:xai_grok_4_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_model_card", "model_reports:zai_glm_5_paper"], "document_share": 0.014061207609594707, "domain": "math", "name": "AIME", "organization_count": 8, "organizations": ["Ai2", "Anthropic", "DeepSeek", "Google", "OpenAI", "Qwen", "Z.ai", "xAI"], "rank": 5, "released": "2024-02-01", "source": "model_reports", "url": "https://maa.org/maa-invitational-competitions/"}, {"aliases": ["LiveCodeBench", "LCB"], "benchmark_id": "livecodebench", "caveat": "Time-sliced problem set. A score without its problem window is not comparable to any other score.", "document_count": 17, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_2_5_report", "model_reports:google_gemma_4_model_card", "model_reports:meta_llama_4", "model_reports:mistral_large_3", "model_reports:mistral_medium_3", "model_reports:openai_gpt_5_system_card", "model_reports:openai_o3_o4_mini_system_card", "model_reports:qwen2_5_coder_report", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card", "model_reports:qwen3_technical_report", "model_reports:xai_grok_4_model_card"], "document_share": 0.014061207609594707, "domain": "coding", "name": "LiveCodeBench", "organization_count": 8, "organizations": ["Ai2", "DeepSeek", "Google", "Meta", "Mistral", "OpenAI", "Qwen", "xAI"], "rank": 6, "released": "2024-03-12", "source": "model_reports", "url": "https://livecodebench.github.io/"}, {"aliases": ["MMLU-Pro", "MMLU Pro"], "benchmark_id": "mmlu_pro", "caveat": "Static closed-set multiple choice; contamination risk rises over time.", "document_count": 12, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:anthropic_claude_3_7_sonnet", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemma_4_model_card", "model_reports:meta_llama_3_1", "model_reports:meta_llama_4", "model_reports:mistral_medium_3", "model_reports:qwen3_5_model_card", "model_reports:qwen3_technical_report"], "document_share": 0.009925558312655087, "domain": "knowledge", "name": "MMLU-Pro", "organization_count": 7, "organizations": ["Ai2", "Anthropic", "DeepSeek", "Google", "Meta", "Mistral", "Qwen"], "rank": 7, "released": "2024-06-03", "source": "model_reports", "url": "https://github.com/TIGER-AI-Lab/MMLU-Pro"}, {"aliases": ["BrowseComp"], "benchmark_id": "browsecomp", "caveat": "Live web. The result depends on what the internet contained on the day of the run.", "document_count": 12, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:openai_gpt_5_6_release", "model_reports:openai_gpt_5_6_system_card", "model_reports:openai_gpt_5_system_card", "model_reports:openai_o3_o4_mini_system_card", "model_reports:qwen3_5_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_model_card", "model_reports:zai_glm_5_paper"], "document_share": 0.009925558312655087, "domain": "agent", "name": "BrowseComp", "organization_count": 6, "organizations": ["DeepSeek", "Google", "Moonshot AI", "OpenAI", "Qwen", "Z.ai"], "rank": 8, "released": "2025-04-10", "source": "model_reports", "url": "https://openai.com/index/browsecomp/"}, {"aliases": ["SWE-bench Pro", "SWE Pro", "SWE-Bench Pro (Public)"], "benchmark_id": "swe_bench_pro", "caveat": "Harness and split version must be recorded with any score.", "document_count": 10, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:anthropic_claude_opus_5_system_card", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:qwen3_8_model_card", "model_reports:tencent_hy4_preview", "model_reports:xai_grok_4_5", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card"], "document_share": 0.008271298593879239, "domain": "coding_agent", "name": "SWE-bench Pro", "organization_count": 7, "organizations": ["Anthropic", "DeepSeek", "Google", "Qwen", "Tencent", "Z.ai", "xAI"], "rank": 9, "released": "2025-09-23", "source": "model_reports", "url": "https://github.com/scaleapi/SWE-bench_Pro-os"}, {"aliases": ["MATH-500", "MATH 500", "MATH"], "benchmark_id": "math_500", "caveat": "Largely saturated at the frontier.", "document_count": 10, "document_ids": ["model_reports:anthropic_claude_3_7_sonnet", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:google_gemini_1_5_report", "model_reports:meta_llama_3_1", "model_reports:mistral_large_2", "model_reports:mistral_medium_3", "model_reports:qwen2_5_coder_report", "model_reports:qwen3_technical_report"], "document_share": 0.008271298593879239, "domain": "math", "name": "MATH-500", "organization_count": 6, "organizations": ["Anthropic", "DeepSeek", "Google", "Meta", "Mistral", "Qwen"], "rank": 10, "released": "2021-03-05", "source": "model_reports", "url": "https://github.com/openai/prm800k"}, {"aliases": ["MMMU", "MMMU-Pro"], "benchmark_id": "mmmu", "caveat": "Image preprocessing and chain-of-thought settings affect results.", "document_count": 10, "document_ids": ["model_reports:anthropic_claude_3_7_sonnet", "model_reports:anthropic_claude_4_system_card", "model_reports:google_gemini_1_5_report", "model_reports:google_gemini_2_5_report", "model_reports:meta_llama_4", "model_reports:openai_gpt_4_1", "model_reports:openai_gpt_5_system_card", "model_reports:openai_o3_o4_mini_system_card", "model_reports:qwen3_5_model_card", "model_reports:xai_grok_4_model_card"], "document_share": 0.008271298593879239, "domain": "multimodal", "name": "MMMU", "organization_count": 6, "organizations": ["Anthropic", "Google", "Meta", "OpenAI", "Qwen", "xAI"], "rank": 11, "released": "2023-11-27", "source": "model_reports", "url": "https://mmmu-benchmark.github.io/"}, {"aliases": ["IFEval"], "benchmark_id": "ifeval", "caveat": "Verifiable-constraint subset only; does not measure answer quality.", "document_count": 9, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:anthropic_claude_3_7_sonnet", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:meta_llama_3_1", "model_reports:mistral_medium_3", "model_reports:openai_gpt_4_1", "model_reports:qwen3_5_model_card", "model_reports:qwen3_technical_report"], "document_share": 0.007444168734491315, "domain": "instruction_following", "name": "IFEval", "organization_count": 7, "organizations": ["Ai2", "Anthropic", "DeepSeek", "Meta", "Mistral", "OpenAI", "Qwen"], "rank": 12, "released": "2023-11-14", "source": "model_reports", "url": "https://github.com/google-research/google-research/tree/master/instruction_following_eval"}, {"aliases": ["MMLU"], "benchmark_id": "mmlu", "caveat": "Saturated at the frontier and heavily contaminated. Useful as a historical baseline, not as a ranking signal.", "document_count": 9, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:google_gemini_1_5_report", "model_reports:meta_llama_3_1", "model_reports:mistral_large_2", "model_reports:openai_gpt_4_1", "model_reports:qwen2_5_coder_report"], "document_share": 0.007444168734491315, "domain": "knowledge", "name": "MMLU", "organization_count": 7, "organizations": ["Ai2", "DeepSeek", "Google", "Meta", "Mistral", "OpenAI", "Qwen"], "rank": 13, "released": "2020-09-07", "source": "model_reports", "url": "https://github.com/hendrycks/test"}, {"aliases": ["HumanEval", "HumanEval+"], "benchmark_id": "humaneval", "caveat": "Saturated. Retained because open-weight cards still report it.", "document_count": 9, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:google_gemini_1_5_report", "model_reports:meta_llama_3_1", "model_reports:mistral_large_2", "model_reports:mistral_medium_3", "model_reports:qwen2_5_coder_report", "model_reports:qwen3_technical_report"], "document_share": 0.007444168734491315, "domain": "coding", "name": "HumanEval", "organization_count": 6, "organizations": ["Ai2", "DeepSeek", "Google", "Meta", "Mistral", "Qwen"], "rank": 14, "released": "2021-07-07", "source": "model_reports", "url": "https://github.com/openai/human-eval"}, {"aliases": ["SimpleQA", "SimpleQA Verified"], "benchmark_id": "simpleqa", "caveat": "Knowledge cutoff must be recorded alongside the score.", "document_count": 9, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_2_5_report", "model_reports:mistral_large_3", "model_reports:openai_gpt_5_system_card", "model_reports:openai_o3_o4_mini_system_card"], "document_share": 0.007444168734491315, "domain": "factuality", "name": "SimpleQA", "organization_count": 5, "organizations": ["Ai2", "DeepSeek", "Google", "Mistral", "OpenAI"], "rank": 15, "released": "2024-10-30", "source": "model_reports", "url": "https://openai.com/index/introducing-simpleqa/"}, {"aliases": ["MCP Atlas", "MCP-Atlas", "MCPAtlas", "MCP-Atlas (Public Set)"], "benchmark_id": "mcp_atlas", "caveat": "Public and full splits are both reported under one name; the tool server set is part of the measurement.", "document_count": 8, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_model_card"], "document_share": 0.006617038875103391, "domain": "tool_use", "name": "MCP Atlas", "organization_count": 5, "organizations": ["DeepSeek", "Google", "Moonshot AI", "Tencent", "Z.ai"], "rank": 16, "released": "2025-11-18", "source": "model_reports", "url": "https://huggingface.co/datasets/Sierra-Research/MCP-Atlas"}, {"aliases": ["MRCR", "OpenAI-MRCR", "MRCR v2"], "benchmark_id": "mrcr", "caveat": "Needle position and generation seed change the result.", "document_count": 8, "document_ids": ["model_reports:anthropic_claude_4_system_card", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_2_5_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:google_gemma_4_model_card", "model_reports:openai_gpt_4_1", "model_reports:openai_gpt_5_system_card"], "document_share": 0.006617038875103391, "domain": "long_context", "name": "MRCR", "organization_count": 4, "organizations": ["Anthropic", "DeepSeek", "Google", "OpenAI"], "rank": 17, "released": "2025-04-14", "source": "model_reports", "url": "https://huggingface.co/datasets/openai/mrcr"}, {"aliases": ["Tool Decathlon", "Toolathlon", "Toolathlon-Verified"], "benchmark_id": "tool_decathlon", "caveat": "Aggregates ten heterogeneous tool suites; the per-suite spread is wide.", "document_count": 8, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_pro_0813_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:moonshot_kimi_k3_model_card", "model_reports:qwen3_5_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_model_card"], "document_share": 0.006617038875103391, "domain": "tool_use", "name": "Tool Decathlon", "organization_count": 4, "organizations": ["DeepSeek", "Moonshot AI", "Qwen", "Z.ai"], "rank": 18, "released": "2025-10-28", "source": "model_reports", "url": "https://toolathlon.xyz/"}, {"aliases": ["GSM8K"], "benchmark_id": "gsm8k", "caveat": "Saturated. A legacy baseline for small open models.", "document_count": 7, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:google_gemini_1_5_report", "model_reports:meta_llama_3_1", "model_reports:mistral_large_2", "model_reports:qwen2_5_coder_report"], "document_share": 0.005789909015715467, "domain": "math", "name": "GSM8K", "organization_count": 6, "organizations": ["Ai2", "DeepSeek", "Google", "Meta", "Mistral", "Qwen"], "rank": 19, "released": "2021-10-27", "source": "model_reports", "url": "https://github.com/openai/grade-school-math"}, {"aliases": ["MMMLU", "Multilingual MMLU", "MMLU-ProX"], "benchmark_id": "mmmlu", "caveat": "A language-count average. Two cards reporting \"MMMLU\" may be averaging over different language sets, so the counts must match to compare.", "document_count": 7, "document_ids": ["model_reports:anthropic_claude_4_system_card", "model_reports:deepseek_v4_model_card", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:google_gemma_4_model_card", "model_reports:meta_llama_4", "model_reports:mistral_large_3", "model_reports:qwen3_5_model_card"], "document_share": 0.005789909015715467, "domain": "multilingual", "name": "MMMLU", "organization_count": 6, "organizations": ["Anthropic", "DeepSeek", "Google", "Meta", "Mistral", "Qwen"], "rank": 20, "released": "2024-09-24", "source": "model_reports", "url": "https://huggingface.co/datasets/openai/MMMLU"}, {"aliases": ["GDPval", "GDPval-AA", "GDPval-AA v2"], "benchmark_id": "gdpval", "caveat": "Graded by expert human comparison against real deliverables, and the \"-AA\" variants are run by Artificial Analysis rather than the vendor, so an Elo here is not comparable to a vendor-run pass rate.", "document_count": 7, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:anthropic_claude_opus_5_system_card", "model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:moonshot_kimi_k3_model_card", "model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.005789909015715467, "domain": "professional", "name": "GDPval", "organization_count": 5, "organizations": ["Anthropic", "DeepSeek", "Moonshot AI", "Tencent", "Z.ai"], "rank": 21, "released": "2025-09-25", "source": "model_reports", "url": "https://openai.com/index/gdpval/"}, {"aliases": ["AutomationBench", "Zapier AutomationBench", "HLEAutomationBench"], "benchmark_id": "automationbench", "caveat": "Published by Zapier over its own automation surface, and cards report different task subsets of it.", "document_count": 6, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:anthropic_claude_opus_5_system_card", "model_reports:deepseek_v4_pro_0813_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.004962779156327543, "domain": "agent", "name": "AutomationBench", "organization_count": 5, "organizations": ["Anthropic", "DeepSeek", "Moonshot AI", "Tencent", "Z.ai"], "rank": 22, "released": "2026-03-10", "source": "model_reports", "url": "https://github.com/zapier/automation-bench"}, {"aliases": ["DROP"], "benchmark_id": "drop", "caveat": "Legacy reading-comprehension baseline.", "document_count": 6, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:google_gemini_1_5_report", "model_reports:meta_llama_3_1"], "document_share": 0.004962779156327543, "domain": "reasoning", "name": "DROP", "organization_count": 4, "organizations": ["Ai2", "DeepSeek", "Google", "Meta"], "rank": 23, "released": "2019-03-01", "source": "model_reports", "url": "https://allenai.org/data/drop"}, {"aliases": ["HMMT", "HMMT Feb 25", "HMMT Nov 25", "HMMT 2026 Feb"], "benchmark_id": "hmmt", "caveat": "A dated competition, not a fixed set: each sitting is a different instrument and contamination rises once problems are published.", "document_count": 6, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:qwen3_5_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_model_card"], "document_share": 0.004962779156327543, "domain": "math", "name": "HMMT", "organization_count": 3, "organizations": ["DeepSeek", "Qwen", "Z.ai"], "rank": 24, "released": "2025-02-15", "source": "model_reports", "url": "https://www.hmmt.org/"}, {"aliases": ["IMOAnswerBench", "IMO AnswerBench"], "benchmark_id": "imo_answer_bench", "caveat": "Final-answer grading on olympiad problems, so it does not check whether the proof reasoning was valid.", "document_count": 6, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:qwen3_5_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_2_model_card", "model_reports:zai_glm_5_model_card"], "document_share": 0.004962779156327543, "domain": "math", "name": "IMOAnswerBench", "organization_count": 3, "organizations": ["DeepSeek", "Qwen", "Z.ai"], "rank": 25, "released": "2025-09-18", "source": "model_reports", "url": "https://huggingface.co/datasets/HuggingFaceH4/IMOAnswerBench"}, {"aliases": ["tau2-bench", "τ²-Bench", "TAU2-Bench", "Tau2", "τ2-bench"], "benchmark_id": "tau2_bench", "caveat": "Per-domain scores (Retail, Telecom, Airline) diverge sharply and several cards apply the Airline domain fixes from the Claude Opus 4.5 system card, so an average conceals both the domain mix and the patch level.", "document_count": 6, "document_ids": ["model_reports:google_gemini_3_1_pro_model_card", "model_reports:google_gemma_4_model_card", "model_reports:qwen3_5_model_card", "model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_model_card", "model_reports:zai_glm_5_paper"], "document_share": 0.004962779156327543, "domain": "tool_use", "name": "tau2-bench", "organization_count": 3, "organizations": ["Google", "Qwen", "Z.ai"], "rank": 26, "released": "2025-06-09", "source": "model_reports", "url": "https://arxiv.org/abs/2506.07982"}, {"aliases": ["MBPP", "MBPP+"], "benchmark_id": "mbpp", "caveat": "Saturated; short single-function tasks.", "document_count": 5, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_v3_report", "model_reports:meta_llama_3_1", "model_reports:mistral_large_2", "model_reports:qwen2_5_coder_report"], "document_share": 0.0041356492969396195, "domain": "coding", "name": "MBPP", "organization_count": 5, "organizations": ["Ai2", "DeepSeek", "Meta", "Mistral", "Qwen"], "rank": 27, "released": "2021-08-16", "source": "model_reports", "url": "https://github.com/google-research/google-research/tree/master/mbpp"}, {"aliases": ["APEX-Agents", "APEX", "Apex", "Apex Shortlist"], "benchmark_id": "apex_agents", "caveat": "Expert-authored professional tasks with rubric grading; graders are LLMs.", "document_count": 5, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemini_3_1_pro_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:tencent_hy4_preview"], "document_share": 0.0041356492969396195, "domain": "agent", "name": "APEX-Agents", "organization_count": 4, "organizations": ["DeepSeek", "Google", "Moonshot AI", "Tencent"], "rank": 28, "released": "2026-01-20", "source": "model_reports", "url": "https://arxiv.org/abs/2601.14242"}, {"aliases": ["SWE-bench Multilingual", "SWE Multilingual"], "benchmark_id": "swe_bench_multilingual", "caveat": "Per-language resolution rates differ widely, so an average hides which languages the model actually handles.", "document_count": 5, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:qwen3_5_model_card", "model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_model_card"], "document_share": 0.0041356492969396195, "domain": "coding_agent", "name": "SWE-bench Multilingual", "organization_count": 4, "organizations": ["DeepSeek", "Qwen", "Tencent", "Z.ai"], "rank": 29, "released": "2025-04-22", "source": "model_reports", "url": "https://www.swebench.com/multilingual.html"}, {"aliases": ["OSWorld", "OSWorld-Verified", "OSWorld 2.0"], "benchmark_id": "osworld", "caveat": "Screen resolution, OS image and action space all change results. The Verified and 2.0 revisions are not comparable to the original.", "document_count": 5, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:anthropic_claude_opus_5_system_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card"], "document_share": 0.0041356492969396195, "domain": "computer_use", "name": "OSWorld", "organization_count": 3, "organizations": ["Anthropic", "Moonshot AI", "Qwen"], "rank": 30, "released": "2024-04-11", "source": "model_reports", "url": "https://os-world.github.io/"}, {"aliases": ["MMLU-Redux"], "benchmark_id": "mmlu_redux", "caveat": "Error-corrected MMLU subset; reported mainly by open-weight cards.", "document_count": 5, "document_ids": ["model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_technical_report"], "document_share": 0.0041356492969396195, "domain": "knowledge", "name": "MMLU-Redux", "organization_count": 2, "organizations": ["DeepSeek", "Qwen"], "rank": 31, "released": "2024-06-06", "source": "model_reports", "url": "https://github.com/aryopg/mmlu-redux"}, {"aliases": ["Aider Polyglot", "Aider"], "benchmark_id": "aider_polyglot", "caveat": "Edit-format sensitive; whole-file and diff modes differ substantially.", "document_count": 4, "document_ids": ["model_reports:deepseek_r1_report", "model_reports:google_gemini_2_5_report", "model_reports:mistral_medium_3", "model_reports:openai_gpt_4_1"], "document_share": 0.0033085194375516956, "domain": "coding", "name": "Aider Polyglot", "organization_count": 4, "organizations": ["DeepSeek", "Google", "Mistral", "OpenAI"], "rank": 32, "released": "2024-12-21", "source": "model_reports", "url": "https://aider.chat/docs/leaderboards/"}, {"aliases": ["PostTrainBench", "PostTrainBench Lite"], "benchmark_id": "posttrainbench", "caveat": "Measures a model's ability to run post-training itself. Hardware differs between reported runs (H100 vs H20), which moves the score.", "document_count": 4, "document_ids": ["model_reports:moonshot_kimi_k3_model_card", "model_reports:openai_gpt_5_6_system_card", "model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_2_model_card"], "document_share": 0.0033085194375516956, "domain": "ai_research", "name": "PostTrainBench", "organization_count": 4, "organizations": ["Moonshot AI", "OpenAI", "Tencent", "Z.ai"], "rank": 33, "released": "2026-01-20", "source": "model_reports", "url": "https://posttrainbench.com/"}, {"aliases": ["Video-MME", "VideoMME"], "benchmark_id": "video_mme", "caveat": "Frame sampling rate dominates long-video results.", "document_count": 4, "document_ids": ["model_reports:google_gemini_2_5_report", "model_reports:moonshot_kimi_k3_model_card", "model_reports:openai_gpt_4_1", "model_reports:qwen3_5_model_card"], "document_share": 0.0033085194375516956, "domain": "multimodal", "name": "Video-MME", "organization_count": 4, "organizations": ["Google", "Moonshot AI", "OpenAI", "Qwen"], "rank": 34, "released": "2024-05-31", "source": "model_reports", "url": "https://video-mme.github.io/"}, {"aliases": ["AGIEval"], "benchmark_id": "agieval", "caveat": "Human-exam derived; overlaps heavily with MMLU-style coverage.", "document_count": 4, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:deepseek_v3_report", "model_reports:deepseek_v4_model_card", "model_reports:meta_llama_3_1"], "document_share": 0.0033085194375516956, "domain": "knowledge", "name": "AGIEval", "organization_count": 3, "organizations": ["Ai2", "DeepSeek", "Meta"], "rank": 35, "released": "2023-04-13", "source": "model_reports", "url": "https://github.com/ruixiangcui/AGIEval"}, {"aliases": ["Arena-Hard", "ArenaHard"], "benchmark_id": "arena_hard", "caveat": "LLM-judge dependent, with known style and length bias.", "document_count": 4, "document_ids": ["model_reports:deepseek_r1_report", "model_reports:deepseek_v3_report", "model_reports:meta_llama_3_1", "model_reports:qwen3_technical_report"], "document_share": 0.0033085194375516956, "domain": "human_preference", "name": "Arena-Hard", "organization_count": 3, "organizations": ["DeepSeek", "Meta", "Qwen"], "rank": 36, "released": "2024-04-19", "source": "model_reports", "url": "https://github.com/lmarena/arena-hard-auto"}, {"aliases": ["BFCL", "Berkeley Function Calling Leaderboard"], "benchmark_id": "bfcl", "caveat": "Schema complexity and execution checking vary by version.", "document_count": 4, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:meta_llama_3_1", "model_reports:qwen3_5_model_card", "model_reports:qwen3_technical_report"], "document_share": 0.0033085194375516956, "domain": "tool_use", "name": "BFCL", "organization_count": 3, "organizations": ["Ai2", "Meta", "Qwen"], "rank": 37, "released": "2024-02-26", "source": "model_reports", "url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}, {"aliases": ["MathVision", "MATH-Vision", "MathVision (mini)"], "benchmark_id": "mathvision", "caveat": "Prompt-format sensitive: vendors report boxed and unboxed variants and sometimes take the higher of the two.", "document_count": 4, "document_ids": ["model_reports:google_gemma_4_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card"], "document_share": 0.0033085194375516956, "domain": "multimodal", "name": "MathVision", "organization_count": 3, "organizations": ["Google", "Moonshot AI", "Qwen"], "rank": 38, "released": "2024-02-22", "source": "model_reports", "url": "https://mathllm.github.io/mathvision/"}, {"aliases": ["MathVista"], "benchmark_id": "mathvista", "caveat": "Mixes OCR quality with visual reasoning.", "document_count": 4, "document_ids": ["model_reports:google_gemini_1_5_report", "model_reports:google_gemini_2_5_report", "model_reports:meta_llama_4", "model_reports:qwen3_5_model_card"], "document_share": 0.0033085194375516956, "domain": "multimodal", "name": "MathVista", "organization_count": 3, "organizations": ["Google", "Meta", "Qwen"], "rank": 39, "released": "2023-10-03", "source": "model_reports", "url": "https://mathvista.github.io/"}, {"aliases": ["MMMU-Pro", "MMMU Pro"], "benchmark_id": "mmmu_pro", "caveat": "Harder vision-required subset of MMMU. Input image ordering changes the result, so the protocol has to travel with the number.", "document_count": 4, "document_ids": ["model_reports:google_gemini_3_1_pro_model_card", "model_reports:google_gemma_4_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:qwen3_5_model_card"], "document_share": 0.0033085194375516956, "domain": "multimodal", "name": "MMMU-Pro", "organization_count": 3, "organizations": ["Google", "Moonshot AI", "Qwen"], "rank": 40, "released": "2024-09-04", "source": "model_reports", "url": "https://arxiv.org/abs/2409.02813"}, {"aliases": ["OmniDocBench", "OmniDocBench1.5", "OmniDocBench 1.5"], "benchmark_id": "omnidocbench", "caveat": "Reported as an edit distance where lower is better, so it inverts the direction of every other row a card puts beside it.", "document_count": 4, "document_ids": ["model_reports:google_gemma_4_model_card", "model_reports:moonshot_kimi_k3_model_card", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card"], "document_share": 0.0033085194375516956, "domain": "multimodal", "name": "OmniDocBench", "organization_count": 3, "organizations": ["Google", "Moonshot AI", "Qwen"], "rank": 41, "released": "2024-12-10", "source": "model_reports", "url": "https://github.com/opendatalab/OmniDocBench"}, {"aliases": ["ARC-AGI", "ARC-AGI-1", "ARC-AGI-2"], "benchmark_id": "arc_agi", "caveat": "Compute per task is reported alongside score by the maintainers and should not be dropped.", "document_count": 3, "document_ids": ["model_reports:google_gemini_2_5_report", "model_reports:openai_o3_o4_mini_system_card", "model_reports:xai_grok_4_model_card"], "document_share": 0.0024813895781637717, "domain": "reasoning", "name": "ARC-AGI", "organization_count": 3, "organizations": ["Google", "OpenAI", "xAI"], "rank": 42, "released": "2019-11-05", "source": "model_reports", "url": "https://arcprize.org/"}, {"aliases": ["CritPt"], "benchmark_id": "critpt", "caveat": "Research-level physics reasoning; small expert-authored set.", "document_count": 3, "document_ids": ["model_reports:moonshot_kimi_k3_model_card", "model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_2_model_card"], "document_share": 0.0024813895781637717, "domain": "science", "name": "CritPt", "organization_count": 3, "organizations": ["Moonshot AI", "Tencent", "Z.ai"], "rank": 43, "released": "2025-09-30", "source": "model_reports", "url": "https://arxiv.org/abs/2509.26574"}, {"aliases": ["GPQA main", "GPQA full"], "benchmark_id": "gpqa", "caveat": "The full split, distinct from the Diamond subset most frontier cards report.", "document_count": 3, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:meta_llama_3_1", "model_reports:qwen3_5_model_card"], "document_share": 0.0024813895781637717, "domain": "science", "name": "GPQA (full)", "organization_count": 3, "organizations": ["Ai2", "Meta", "Qwen"], "rank": 44, "released": "2023-11-20", "source": "model_reports", "url": "https://arxiv.org/abs/2311.12022"}, {"aliases": ["Codeforces", "Codeforces ELO", "Codeforces Rating"], "benchmark_id": "codeforces", "caveat": "An Elo estimate against human contestants, not a fixed dataset. The problem set moves continuously and rating conversion differs by vendor.", "document_count": 3, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:google_gemma_4_model_card"], "document_share": 0.0024813895781637717, "domain": "coding", "name": "Codeforces", "organization_count": 2, "organizations": ["DeepSeek", "Google"], "rank": 45, "released": "2010-02-19", "source": "model_reports", "url": "https://codeforces.com/"}, {"aliases": ["IFBench"], "benchmark_id": "ifbench", "caveat": "Verifiable constraints on unseen instruction types; does not measure answer quality.", "document_count": 3, "document_ids": ["model_reports:ai2_olmo_3", "model_reports:qwen3_5_model_card", "model_reports:qwen3_8_model_card"], "document_share": 0.0024813895781637717, "domain": "instruction_following", "name": "IFBench", "organization_count": 2, "organizations": ["Ai2", "Qwen"], "rank": 46, "released": "2025-07-03", "source": "model_reports", "url": "https://arxiv.org/abs/2507.02833"}, {"aliases": ["LongBench", "LongBench v2"], "benchmark_id": "longbench", "caveat": "Nominal context length is not effective context length.", "document_count": 3, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:deepseek_v4_technical_report", "model_reports:qwen3_5_model_card"], "document_share": 0.0024813895781637717, "domain": "long_context", "name": "LongBench", "organization_count": 2, "organizations": ["DeepSeek", "Qwen"], "rank": 47, "released": "2023-08-28", "source": "model_reports", "url": "https://github.com/THUDM/LongBench"}, {"aliases": ["tau-bench", "τ-bench", "tau2-bench", "TAU-bench"], "benchmark_id": "tau_bench", "caveat": "Depends on a simulated user and policy; the simulator model is part of the measurement.", "document_count": 3, "document_ids": ["model_reports:anthropic_claude_3_7_sonnet", "model_reports:anthropic_claude_4_system_card", "model_reports:openai_gpt_5_system_card"], "document_share": 0.0024813895781637717, "domain": "tool_use", "name": "tau-bench", "organization_count": 2, "organizations": ["Anthropic", "OpenAI"], "rank": 48, "released": "2024-06-17", "source": "model_reports", "url": "https://github.com/sierra-research/tau-bench"}, {"aliases": ["AA-LCR"], "benchmark_id": "aa_lcr", "caveat": "Run by Artificial Analysis rather than the vendor. Third-party execution is the point, but it also means the vendor did not control the setup.", "document_count": 2, "document_ids": ["model_reports:moonshot_kimi_k3_model_card", "model_reports:qwen3_5_model_card"], "document_share": 0.0016542597187758478, "domain": "long_context", "name": "AA-LCR", "organization_count": 2, "organizations": ["Moonshot AI", "Qwen"], "rank": 49, "released": "2025-09-16", "source": "model_reports", "url": "https://artificialanalysis.ai/evaluations/aa-lcr"}, {"aliases": ["Agents' Last Exam", "ALE", "Agent's Last Exam", "ALE-CLI"], "benchmark_id": "agents_last_exam", "caveat": "Multi-step agentic benchmark using Claude Code harness. Tool Search disabled. Scores depend on harness, reasoning effort, context length, and timeout settings.", "document_count": 2, "document_ids": ["model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0016542597187758478, "domain": "coding_agent", "name": "Agents' Last Exam", "organization_count": 2, "organizations": ["Tencent", "Z.ai"], "rank": 50, "released": "2025-01-23", "source": "model_reports", "url": "https://lastexam.ai/"}, {"aliases": ["ARC-AGI-2", "ARC-AGI 2"], "benchmark_id": "arc_agi_2", "caveat": "Cost per task is part of the official result and is routinely dropped when the score is quoted on its own.", "document_count": 2, "document_ids": ["model_reports:google_gemini_3_1_pro_model_card", "model_reports:xai_grok_4_5"], "document_share": 0.0016542597187758478, "domain": "reasoning", "name": "ARC-AGI-2", "organization_count": 2, "organizations": ["Google", "xAI"], "rank": 51, "released": "2025-03-24", "source": "model_reports", "url": "https://arcprize.org/arc-agi/2/"}, {"aliases": ["BioMysteryBench"], "benchmark_id": "biomysterybench", "caveat": "Reported in two splits (\"hard\" and \"human solved\") that differ by more than 35 points, so a bare score is unreadable without its split. Anthropic notes its own safety refusals depress this number, which means the score mixes capability with policy. First-party to Anthropic, which built it and publishes the dataset. The task set was revised after an answer-key audit, so the item count moved and scores are not comparable across dataset versions.", "document_count": 2, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:tencent_hy4_preview"], "document_share": 0.0016542597187758478, "domain": "biology", "name": "BioMysteryBench", "organization_count": 2, "organizations": ["Anthropic", "Tencent"], "rank": 52, "released": "2026-04-29", "source": "model_reports", "url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-full"}, {"aliases": ["BrowseComp-ZH", "BrowseComp-zh", "BrowseComp-Zh"], "benchmark_id": "browsecomp_zh", "caveat": "Chinese-language live web. Same day-to-day web drift as BrowseComp, over a different index.", "document_count": 2, "document_ids": ["model_reports:qwen3_5_model_card", "model_reports:zai_glm_5_model_card"], "document_share": 0.0016542597187758478, "domain": "agent", "name": "BrowseComp-ZH", "organization_count": 2, "organizations": ["Qwen", "Z.ai"], "rank": 53, "released": "2025-04-27", "source": "model_reports", "url": "https://arxiv.org/abs/2504.19314"}, {"aliases": ["ChartQA"], "benchmark_id": "chartqa", "caveat": "Largely saturated; relaxed-accuracy tolerance affects the reported figure.", "document_count": 2, "document_ids": ["model_reports:google_gemini_1_5_report", "model_reports:meta_llama_4"], "document_share": 0.0016542597187758478, "domain": "multimodal", "name": "ChartQA", "organization_count": 2, "organizations": ["Google", "Meta"], "rank": 54, "released": "2022-03-19", "source": "model_reports", "url": "https://github.com/vis-nlp/ChartQA"}, {"aliases": ["DeepSearchQA"], "benchmark_id": "deepsearchqa", "caveat": "Live-web deep research; F1 grading depends on the reference answer set.", "document_count": 2, "document_ids": ["model_reports:anthropic_claude_opus_5_system_card", "model_reports:moonshot_kimi_k3_model_card"], "document_share": 0.0016542597187758478, "domain": "agent", "name": "DeepSearchQA", "organization_count": 2, "organizations": ["Anthropic", "Moonshot AI"], "rank": 55, "released": "2026-02-10", "source": "model_reports", "url": "https://huggingface.co/datasets/PokeeAI/DeepSearchQA"}, {"aliases": ["DeepSWE"], "benchmark_id": "deepswe", "caveat": "Mini-swe-agent harness, temperature=0.95, top_p=1.0, max_new_tokens=64k under 1M context.", "document_count": 2, "document_ids": ["model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0016542597187758478, "domain": "coding_agent", "name": "DeepSWE", "organization_count": 2, "organizations": ["Tencent", "Z.ai"], "rank": 56, "released": "2025-09-15", "source": "model_reports", "url": "https://github.com/deepswe/deepswe"}, {"aliases": ["DocVQA"], "benchmark_id": "docvqa", "caveat": "Saturated at the frontier; ANLS scoring is lenient to OCR near-misses.", "document_count": 2, "document_ids": ["model_reports:google_gemini_1_5_report", "model_reports:meta_llama_4"], "document_share": 0.0016542597187758478, "domain": "multimodal", "name": "DocVQA", "organization_count": 2, "organizations": ["Google", "Meta"], "rank": 57, "released": "2020-07-01", "source": "model_reports", "url": "https://www.docvqa.org/"}, {"aliases": ["ExploitBench", "ExploitBench (Cap%)"], "benchmark_id": "exploitbench", "caveat": "Offensive-security capability measured as a risk indicator against a preparedness threshold, not a leaderboard to top. Reported as a capability percentage. Published by Seunghyun Lee and David Brumley (Carnegie Mellon; Brumley also lists Bugcrowd) as arXiv:2605.14153 with code at github.com/exploitbench/exploitbench. 41 V8 N-day vulnerabilities scored by 16 grader-verified capability flags; vendor runs report the capability percentage without publishing per-flag results, so a reported number is not reproducible from the paper alone.", "document_count": 2, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:openai_gpt_5_6_system_card"], "document_share": 0.0016542597187758478, "domain": "security", "name": "ExploitBench", "organization_count": 2, "organizations": ["Anthropic", "OpenAI"], "rank": 58, "released": "2026-05-13", "source": "model_reports", "url": "https://exploitbench.ai/"}, {"aliases": ["MCPMark", "MCP-Mark", "MCPMark-Verified"], "benchmark_id": "mcp_mark", "caveat": "Depends on live third-party MCP servers, so runs are not reproducible over time.", "document_count": 2, "document_ids": ["model_reports:moonshot_kimi_k3_model_card", "model_reports:qwen3_5_model_card"], "document_share": 0.0016542597187758478, "domain": "tool_use", "name": "MCPMark", "organization_count": 2, "organizations": ["Moonshot AI", "Qwen"], "rank": 59, "released": "2025-09-30", "source": "model_reports", "url": "https://mcpmark.ai/"}, {"aliases": ["NL2Repo", "NL-to-Repo"], "benchmark_id": "nl2repo", "caveat": "Evaluates generating a repository from a natural language description. Temperature=1.0, top_p=1.0, max_new_tokens=64k under 1M context.", "document_count": 2, "document_ids": ["model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0016542597187758478, "domain": "coding_agent", "name": "NL2Repo", "organization_count": 2, "organizations": ["Tencent", "Z.ai"], "rank": 60, "released": "2025-06-15", "source": "model_reports", "url": "https://github.com/nl2repo/nl2repo"}, {"aliases": ["OfficeQA Pro", "OfficeQA"], "benchmark_id": "office_qa_pro", "caveat": "Visual QA benchmark on professional document corpus (PDF without embedded text). Scores depend on context length and image resolution.", "document_count": 2, "document_ids": ["model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0016542597187758478, "domain": "vision", "name": "OfficeQA Pro", "organization_count": 2, "organizations": ["Tencent", "Z.ai"], "rank": 61, "released": "2025-06-01", "source": "model_reports", "url": "https://github.com/OfficeQA/OfficeQA-Pro"}, {"aliases": ["SciCode"], "benchmark_id": "scicode", "caveat": "Scientific code generation graded by test execution; subproblem context matters.", "document_count": 2, "document_ids": ["model_reports:google_gemini_3_1_pro_model_card", "model_reports:moonshot_kimi_k3_model_card"], "document_share": 0.0016542597187758478, "domain": "science", "name": "SciCode", "organization_count": 2, "organizations": ["Google", "Moonshot AI"], "rank": 62, "released": "2024-07-18", "source": "model_reports", "url": "https://scicode-bench.github.io/"}, {"aliases": ["SuperGPQA", "Super-GPQA"], "benchmark_id": "super_gpqa", "caveat": "285 graduate disciplines; breadth comes at the cost of per-field sample size.", "document_count": 2, "document_ids": ["model_reports:deepseek_v4_model_card", "model_reports:qwen3_5_model_card"], "document_share": 0.0016542597187758478, "domain": "knowledge", "name": "SuperGPQA", "organization_count": 2, "organizations": ["DeepSeek", "Qwen"], "rank": 63, "released": "2025-02-20", "source": "model_reports", "url": "https://arxiv.org/abs/2502.14739"}, {"aliases": ["Toolathlon Verified", "Toolathlon"], "benchmark_id": "toolathlon_verified", "caveat": "Pass@1 averaged over 3 independent runs. Official evaluation service.", "document_count": 2, "document_ids": ["model_reports:tencent_hy4_preview", "model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0016542597187758478, "domain": "tool_use", "name": "Toolathlon Verified", "organization_count": 2, "organizations": ["Tencent", "Z.ai"], "rank": 64, "released": "2025-11-01", "source": "model_reports", "url": "https://github.com/toolathlon/toolathlon"}, {"aliases": ["WideSearch"], "benchmark_id": "widesearch", "caveat": "Wide-coverage web collection; scored on completeness of an enumerated answer set.", "document_count": 2, "document_ids": ["model_reports:qwen3_5_model_card", "model_reports:tencent_hy4_preview"], "document_share": 0.0016542597187758478, "domain": "agent", "name": "WideSearch", "organization_count": 2, "organizations": ["Qwen", "Tencent"], "rank": 65, "released": "2025-08-11", "source": "model_reports", "url": "https://widesearch-seed.github.io/"}, {"aliases": ["CursorBench", "Cursor Bench", "CursorBench 3.2"], "benchmark_id": "cursor_bench", "caveat": "Maintained by Cursor and measured inside the Cursor product, so it reflects that scaffold rather than the bare model.", "document_count": 2, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5", "model_reports:anthropic_claude_opus_5_system_card"], "document_share": 0.0016542597187758478, "domain": "coding_agent", "name": "CursorBench", "organization_count": 1, "organizations": ["Anthropic"], "rank": 66, "released": "2025-10-29", "source": "model_reports", "url": "https://cursor.com/blog/cursor-bench"}, {"aliases": ["HealthBench", "HealthBench Hard", "HealthBench Consensus"], "benchmark_id": "healthbench", "caveat": "Physician-written rubrics scored by an LLM grader, and the length-adjusted variant is what recent cards report. Published by OpenAI.", "document_count": 2, "document_ids": ["model_reports:openai_gpt_5_6_system_card", "model_reports:openai_gpt_5_system_card"], "document_share": 0.0016542597187758478, "domain": "health", "name": "HealthBench", "organization_count": 1, "organizations": ["OpenAI"], "rank": 67, "released": "2025-05-12", "source": "model_reports", "url": "https://openai.com/index/healthbench/"}, {"aliases": ["MLE-Bench", "MLE-bench Revised", "MLE-bench Lite"], "benchmark_id": "mle_bench", "caveat": "Kaggle-derived ML engineering tasks; wall-clock budget and hardware are part of the result. Published by OpenAI.", "document_count": 2, "document_ids": ["model_reports:openai_gpt_5_6_system_card", "model_reports:openai_gpt_5_system_card"], "document_share": 0.0016542597187758478, "domain": "ai_research", "name": "MLE-bench", "organization_count": 1, "organizations": ["OpenAI"], "rank": 68, "released": "2024-10-09", "source": "model_reports", "url": "https://openai.com/index/mle-bench/"}, {"aliases": ["Vending Bench", "Vending Bench 2"], "benchmark_id": "vending_bench", "caveat": "Reported in simulated dollars, not a percentage, and run over long horizons where a single early failure dominates the total.", "document_count": 2, "document_ids": ["model_reports:zai_glm_5_1_model_card", "model_reports:zai_glm_5_model_card"], "document_share": 0.0016542597187758478, "domain": "agent", "name": "Vending Bench", "organization_count": 1, "organizations": ["Z.ai"], "rank": 69, "released": "2025-02-18", "source": "model_reports", "url": "https://andonlabs.com/evals/vending-bench-2"}, {"aliases": [], "benchmark_id": "opencompass-1580-a-bench", "caveat": "A-Bench is a benchmark designed to diagnose whether LMMs are masters at evaluating AIGIs. 2,864 AIGIs from 16 text-to-image models are sampled, each paired with question-answers annotated by human experts, and tested across 18 leading LMMs. A-Bench是一个旨在诊断 LMMs 是否擅长评估 AIGIs 的基准，从 16 个文本到图像模型中采样了 2,864 个 AIGIs，每个都与由人类专家标注的问题-答案配对。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/A-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "A-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 70, "released": "2024-06-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/A-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1367-a-okvqa", "caveat": "A-OKVQA assesses commonsense reasoning abilities. It is a crowdsourced dataset composed of a diverse set of about 25K questions requiring a broad base of commonsense and world knowledge to answer. A-OKVQA用于评估多模态大模型的常识及推理能力，由25K个不同的问题组成，需要对图像中描述的场景进行某种形式的常识性推理来回答。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/A-OKVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "A-OKVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 71, "released": "2022-06-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/A-OKVQA"}, {"aliases": [], "benchmark_id": "artificial-analysis-aa-analystagent", "caveat": "Quantitative analysis on spreadsheets & documents", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/aa-analyst-agent"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "AA-AnalystAgent", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 72, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/aa-analyst-agent"}, {"aliases": [], "benchmark_id": "artificial-analysis-aa-briefcase", "caveat": "Agentic knowledge work, Elo", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/aa-briefcase"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "AA-Briefcase", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 73, "released": "2026-06-18", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/aa-briefcase"}, {"aliases": [], "benchmark_id": "llm-stats-aa-briefcase", "caveat": "AA-Briefcase is an Artificial Analysis evaluation of AI systems on professional knowledge-work tasks, reported as an Elo score.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500"], "document_share": 0.0008271298593879239, "domain": "productivity", "name": "AA-Briefcase", "organization_count": 1, "organizations": ["llm_stats"], "rank": 74, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-briefcase?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-aa-index", "caveat": "No official academic documentation found for this benchmark. Extensive research through ArXiv, IEEE/ACL/NeurIPS papers, and university research sites yielded no peer-reviewed sources for an 'aa-index' benchmark. This entry requires verification from official academic sources.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "AA-Index", "organization_count": 1, "organizations": ["llm_stats"], "rank": 75, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-index?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-aa-lcr", "caveat": "Long context reasoning", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "AA-LCR", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 76, "released": "2025-08-05", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning"}, {"aliases": [], "benchmark_id": "llm-stats-aa-lcr", "caveat": "Agent Arena Long Context Reasoning benchmark", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "AA-LCR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 77, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-lcr?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-aa-omniscience-accuracy", "caveat": "Knowledge", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/omniscience"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "AA-Omniscience Accuracy", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 78, "released": "2025-11-16", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/omniscience"}, {"aliases": [], "benchmark_id": "llm-stats-aa-omniscience-index", "caveat": "AA-Omniscience Index is Artificial Analysis's knowledge-reliability metric. It rewards correct answers, penalizes hallucinations, and does not penalize abstention. Scores range from -100 to 100, where 0 means as many correct as incorrect answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AA-Omniscience Index", "organization_count": 1, "organizations": ["llm_stats"], "rank": 79, "released": "2025-11-16", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aa-omniscience-index?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-aa-omniscience-non-hallucination", "caveat": "1 - hallucination rate", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/omniscience"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "AA-Omniscience Non-Hallucination Rate", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 80, "released": "2025-11-16", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/omniscience"}, {"aliases": [], "benchmark_id": "opencompass-1148-abspyramid", "caveat": "ABSPYRAMID is a unified entailment graph of 221K textual descriptions of abstraction knowledge. ABSPYRAMID collects abstract knowledge for three components\nof diverse events to comprehensively evaluate the abstraction ability of language\nmodels in the open domain. ABSPYRAMID 是包含 221,000 条文本描述的抽象知识,收集了多种事件的三个组成部分的抽象知识，以全面评估语言模型在开放域中的抽象能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AbsPyramid"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "AbsPyramid", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 81, "released": "2024-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AbsPyramid"}, {"aliases": [], "benchmark_id": "llm-stats-acebench", "caveat": "ACEBench is a comprehensive benchmark for evaluating Large Language Models' tool usage capabilities across three primary evaluation types: Normal (basic tool usage scenarios), Special (tool usage with ambiguous or incomplete instructions), and Agent (multi-agent interactions simulating real-world dialogues). The benchmark covers 4,538 APIs across 8 major domains and 68 sub-domains including technology, finance, entertainment, society, health, culture, and environment, supporting both English and Chinese languages.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ACEBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 82, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/acebench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1319-actionatlas", "caveat": "ActionAtlas is a multiple-choice video question answering benchmark, including 934 videos showcasing 580 unique actions across 56 sports, with a total of 1896 actions within choices. ActionAtlas是一个多项选择视频问答基准测试，包括934个视频，展示了56项运动中的580个独特动作，选项共包含1896个动作。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ActionAtlas"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ActionAtlas", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 83, "released": "2024-10-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ActionAtlas"}, {"aliases": [], "benchmark_id": "llm-stats-activitynet", "caveat": "A large-scale video benchmark for human activity understanding. Provides samples from 203 activity classes with an average of 137 untrimmed videos per class and 1.41 activity instances per video, for a total of 849 video hours. The benchmark covers a wide range of complex human activities that are of interest to people in their daily living and can be used to compare algorithms for three scenarios: untrimmed video classification, trimmed activity classification, and activity detection.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500"], "document_share": 0.0008271298593879239, "domain": "video", "name": "ActivityNet", "organization_count": 1, "organizations": ["llm_stats"], "rank": 84, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/activitynet?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1155-ada-leval", "caveat": "Ada-LEval is a length-adaptable benchmark for evaluating the long-context understanding\nof LLMs. Ada-LEval includes two challenging subsets, TSort and BestAnswer, which enable\na more reliable evaluation of LLMs’ long context capabilities. Ada-LEval 用于评估大型语言模型（LLMs）对长上下文的理解能力。Ada-LEval 包含两个具有挑战性的子集，TSort 和 BestAnswer，能够更可靠地评估 LLMs 的长上下文能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Ada-LEval"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "Ada-LEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 85, "released": "2024-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Ada-LEval"}, {"aliases": [], "benchmark_id": "llm-stats-advancedif", "caveat": "AdvancedIF is a rubric-based benchmark measuring complex, multi-turn, and system-prompted instruction following ability, scored with a calibrated LLM judge against per-instruction rubrics.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AdvancedIF", "organization_count": 1, "organizations": ["llm_stats"], "rank": 86, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/advancedif?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2452-aecbench", "caveat": "AECBench is an open-source benchmark for evaluating LLMs in architecture, engineering, and construction (AEC), covering 23 tasks and about 4,800 samples across five cognitive levels. AECBench 是面向建筑、工程与施工（AEC）领域的大语言模型评测基准，覆盖 5 个认知层级、23 类任务和约 4,800 个样本，用于评估模型在知识记忆、理解、推理、计算与应用方面的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AECBench"], "document_share": 0.0008271298593879239, "domain": "科学", "name": "AECBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 87, "released": "2026-04-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AECBench"}, {"aliases": [], "benchmark_id": "llm-stats-community-5f95f778-c521-43fa-b80e-6a55465601e3", "caveat": null, "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A5f95f778-c521-43fa-b80e-6a55465601e3?top_n=500"], "document_share": 0.0008271298593879239, "domain": "other", "name": "ael_gate_benchmark_cases_template", "organization_count": 1, "organizations": ["llm_stats"], "rank": 88, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A5f95f778-c521-43fa-b80e-6a55465601e3?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-aethercode", "caveat": "AetherCode is a competitive-programming benchmark of olympiad-level algorithmic coding problems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AetherCode", "organization_count": 1, "organizations": ["llm_stats"], "rank": 89, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aethercode?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-506-afqmc", "caveat": "AFQMC is an Ant Financial chinese semantic similarity task, which requires to judge whether two sentences have the same meaning or not. AFQMC一个蚂蚁金服中文语义相似度任务，要求判断两个句子是否具有相同的语义。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AFQMC"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "AFQMC", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 90, "released": "2020-04-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AFQMC"}, {"aliases": [], "benchmark_id": "llm-stats-agent-startup-bench", "caveat": "Agent Startup Bench measures AI agents on high-economic-value, startup-style tasks that require autonomous planning and execution to deliver practical, verifiable results.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Agent Startup Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 91, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/agent-startup-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1242-agentboard", "caveat": "AgentBoard is tailored to analytical evaluation of LLM agents. It offers a fine-grained progress rate metric that captures incremental advancements as well as a comprehensive evaluation toolkit that features easy assessment of agents for multi-faceted analysis through interactive visualization. AgentBoard专用于LLM Agent的分析评估，它提供了一个精细指标用于捕获增量进步，以及一个全面的评估工具包，能基于交互式可视化评估进行多方面分析。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentBoard"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "AgentBoard", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 92, "released": "2024-06-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentBoard"}, {"aliases": [], "benchmark_id": "opencompass-1351-agentharm", "caveat": "AgentHarm tests the robustness of LLMs to jailbreak attacks. It includes a diverse set of 110 explicitly malicious agent tasks (440 with augmentations), covering 11 harm categories including fraud, cybercrime, and harassment. AgentHarm用于评估LLM智能体对越狱攻击的鲁棒性，包括110套恶意智能体任务（其中有440个强化任务），涵盖欺诈、网络犯罪和骚扰等11个危害类别。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentHarm"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "AgentHarm", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 93, "released": "2024-10-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentHarm"}, {"aliases": [], "benchmark_id": "opencompass-2061-agenthazard", "caveat": "移动端 GUI Agent 通过与设备环境的交互来完成任务，在完成任务的过程中会遇到一些未知或不可信的信息来源，这些信息可能含有攻击性的内容，致使 Agent 无法正常完成任务，甚至对用户的隐私和财产带来危害。本评测集兼具动态执行环境和静态评测数据集，旨在为移动端 GUI Agent 提供一个仿真度高的模拟环境，以评估其在真实场景下执行的行为和安全性。 移动端 GUI Agent 通过与设备环境的交互来完成任务，在完成任务的过程中会遇到一些未知或不可信的信息来源，这些信息可能含有攻击性的内容，致使 Agent 无法正常完成任务，甚至对用户的隐私和财产带来危害。本评测集兼具动态执行环境和静态评测数据集，旨在为移动端 GUI Agent 提供一个仿真度高的模拟环境，以评估其在真实场景下执行的行为和安全性。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentHazard"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "AgentHazard", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 94, "released": "2025-07-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentHazard"}, {"aliases": [], "benchmark_id": "opencompass-1778-agentrewardbench", "caveat": "AgentRewardBench, the first benchmark to assess the effectiveness of LLM judges for evaluating web agents. AgentRewardBench contains 1302 trajectories across 5 benchmarks and 4 LLMs. AgentRewardBench 是首个用于评估大型语言模型（LLM）评判者评估网络代理有效性的基准测试。AgentRewardBench 包含来自 5 个基准测试和 4 个大型语言模型的 1302 条轨迹。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgentRewardBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "AgentRewardBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 95, "released": "2025-04-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AgentRewardBench"}, {"aliases": [], "benchmark_id": "llm-stats-agents-last-exam", "caveat": "Agents' Last Exam is a challenging benchmark for AI agents on hard, long-horizon tasks that test sustained reasoning, planning, and tool use, reported with and without tool access.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Agents' Last Exam", "organization_count": 1, "organizations": ["llm_stats"], "rank": 96, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/agents-last-exam?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-agieval", "caveat": "A human-centric benchmark for evaluating foundation models on standardized exams including college entrance exams (Gaokao, SAT), law school admission tests (LSAT), math competitions, lawyer qualification tests, and civil service exams. Contains 20 tasks (18 multiple-choice, 2 cloze) designed to assess understanding, knowledge, reasoning, and calculation abilities in real-world academic and professional contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "AGIEval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 97, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/agieval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-497-agieval", "caveat": "AGIEval is a human-centric benchmark specifically designed to evaluate the general abilities of foundation models in tasks pertinent to human cognition and problem-solving. This benchmark is derived from 20 official, public, and high-standard admission and qualification exams intended for general human test-takers, such as general college admission tests (e.g., Chinese College Entrance Exam (Gaokao) and American SAT), law school admission tests, math competitions, lawyer qualification tests, and national civil service exams. AGIEval是一个以人为中心的基准测试，专门设计用于评估基础模型在涉及人类认知和问题解决的任务中的一般能力。该基准测试源自20个官方、公开和高标准的入学和资格考试，例如普通大学入学考试（例如中国高考和美国SAT）、法学院入学考试、数学竞赛、律师资格考试以及国家公务员考试", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AGIEval"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "AGIEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 98, "released": "2023-09-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AGIEval"}, {"aliases": [], "benchmark_id": "opencompass-1753-agmmu", "caveat": "A Comprehensive Agricultural Multimodal Understanding and Reasoning Benchmark 农业综合多模态理解和推理基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AgMMU"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "AgMMU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 99, "released": "2025-04-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AgMMU"}, {"aliases": [], "benchmark_id": "llm-stats-ai2-reasoning-challenge-arc", "caveat": "A dataset of 7,787 genuine grade-school level, multiple-choice science questions assembled to encourage research in advanced question-answering. The dataset is partitioned into a Challenge Set and Easy Set, where the Challenge Set contains only questions answered incorrectly by both retrieval-based and word co-occurrence algorithms. Covers multiple scientific domains including biology, physics, earth science, and chemistry, requiring scientific reasoning, causal understanding, and conceptual knowledge beyond simple fact retrieval. Includes a supporting corpus of over 14 million science sentences.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AI2 Reasoning Challenge (ARC)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 100, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2-reasoning-challenge-%28arc%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ai2d", "caveat": "AI2D is a dataset of 4,903 illustrative diagrams from grade school natural sciences (such as food webs, human physiology, and life cycles) with over 15,000 multiple choice questions and answers. The benchmark evaluates diagram understanding and visual reasoning capabilities, requiring models to interpret diagrammatic elements, relationships, and structure to answer questions about scientific concepts represented in visual form.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "AI2D", "organization_count": 1, "organizations": ["llm_stats"], "rank": 101, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ai2d?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2396-aidabench", "caveat": "AIDABench provides a professional benchmarking suite for AI office data analytics tools, focusing on the core data-processing needs in everyday office workflows, covering high-frequency and reproducible data-processing scenarios commonly seen in real business contexts. AIDABench致力于为AI办公数据分析工具提供专业评测基准，聚焦日常办公中的核心数据处理需求。评测输入以 Excel 文件为主，辅以少量DOC、 PDF、图片及无文件文本等形态，覆盖真实业务中高频、可复现的数据处理场景。核心考察能力包括：数据结构识别、数据清洗、条件筛选、分组聚合、描述统计、排序与排名。任务交付类型包括：问答（Q/A）、文件生成、数据可视化。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIDABench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "AIDABench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 102, "released": "2026-02-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AIDABench"}, {"aliases": [], "benchmark_id": "llm-stats-aider", "caveat": "Aider is a comprehensive code editing benchmark based on 133 practice exercises from Exercism's Python repository, designed to evaluate AI models' ability to translate natural language coding requests into executable code that passes unit tests. The benchmark measures end-to-end code editing capabilities, including GPT's ability to edit existing code and format code changes for automated saving to local files. The Aider Polyglot variant extends this evaluation across 225 challenging exercises spanning C++, Go, Java, JavaScript, Python, and Rust, making it a standard benchmark for assessing multilingual code editing performance in AI research.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Aider", "organization_count": 1, "organizations": ["llm_stats"], "rank": 103, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aider?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-aider-polyglot", "caveat": "A coding benchmark that evaluates LLMs on 225 challenging Exercism programming exercises across C++, Go, Java, JavaScript, Python, and Rust. Models receive two attempts to solve each problem, with test error feedback provided after the first attempt if it fails. The benchmark measures both initial problem-solving ability and capacity to edit code based on error feedback, providing an end-to-end evaluation of code generation and editing capabilities across multiple programming languages.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "Aider-Polyglot", "organization_count": 1, "organizations": ["llm_stats"], "rank": 104, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-aider-polyglot-edit", "caveat": "A challenging multi-language coding benchmark that evaluates models' code editing abilities across C++, Go, Java, JavaScript, Python, and Rust. Contains 225 of Exercism's most difficult programming problems, selected as problems that were solved by 3 or fewer out of 7 top coding models. The benchmark focuses on code editing tasks and measures both correctness of solutions and proper edit format usage. Designed to re-calibrate evaluation scales so top models score between 5-50%.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "Aider-Polyglot Edit", "organization_count": 1, "organizations": ["llm_stats"], "rank": 105, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aider-polyglot-edit?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-aime", "caveat": "American Invitational Mathematics Examination (AIME) benchmark for evaluating mathematical reasoning capabilities of large language models. Contains 30 challenging mathematical problems from AIME 2024 competition that require multi-step reasoning and advanced mathematical insight. Each problem has an integer answer between 000-999.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "AIME", "organization_count": 1, "organizations": ["llm_stats"], "rank": 106, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-aime-2024", "caveat": "American Invitational Mathematics Examination 2024, consisting of 30 challenging mathematical reasoning problems from AIME I and AIME II competitions. Each problem requires an integer answer between 0-999 and tests advanced mathematical reasoning across algebra, geometry, combinatorics, and number theory. Used as a benchmark for evaluating mathematical reasoning capabilities in large language models at Olympiad-level difficulty.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "AIME 2024", "organization_count": 1, "organizations": ["llm_stats"], "rank": 107, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2024?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-aime-2025", "caveat": "Mathematical reasoning", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/aime-2025"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AIME 2025", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 108, "released": "2025-02-12", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/aime-2025"}, {"aliases": [], "benchmark_id": "llm-stats-aime-2025", "caveat": "All 30 problems from the 2025 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "AIME 2025", "organization_count": 1, "organizations": ["llm_stats"], "rank": 109, "released": "2025-02-12", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2025?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-aime-2026", "caveat": "All 30 problems from the 2026 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "AIME 2026", "organization_count": 1, "organizations": ["llm_stats"], "rank": 110, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aime-2026?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-air-bench", "caveat": "AIR-Bench 2024 is a safety benchmark grounded in risk categories derived from government regulations and company policies. It evaluates policy-grounded refusal across a broad regulatory and policy-derived harm taxonomy, using category-specific LLM-judge prompts that reward safe engagement rather than only penalizing unsafe responses.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "AIR-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 111, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/air-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1069-air-bench", "caveat": "AIR-Bench is the first benchmark designed to evaluate the\nability of LALMs to understand various types of audio signals (including human speech, natural sounds, and music), and furthermore, to interact with humans in the textual format. AIR-Bench 是第一个旨在评估 LALMs 理解各种音频信号（包括人类语言、自然声音和音乐）能力的基准，并进一步评估其以文本形式与人类互动的能力。AIR-Bench 包括两个维度：基础基准和聊天基准。前者由19个任务组成，包含约19,000个单选题，旨在检查LALMs的基本单任务能力。后者包含2,000个开放式问答数据实例，直接评估模型对复杂音频的理解及其遵循指令的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIR-Bench"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "AIR-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 112, "released": "2024-02-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AIR-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1557-airbench-2024", "caveat": "AIR-Bench 2024, the first policy-aligned AI safety benchmark, structures 8 government regulations and 16 corporate policies into four security tiers, with 5,694 diverse prompts spanning these categories. AIR-Bench 2024是首个与新兴政府法规和企业政策相一致的 AI 安全基准， 将 8 项政府法规和 16 项企业政策分解为四级安全分类，涵盖了这些类别的 5,694 个多样化的提示。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIRBench-2024"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "AIRBench-2024", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 113, "released": "2024-08-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AIRBench-2024"}, {"aliases": [], "benchmark_id": "opencompass-1995-airtbench", "caveat": "AIRTBench is a benchmark designed to evaluate large language models (LLMs) on their autonomous AI red teaming capabilities. AIRTBench 是一个专为评估大型语言模型（LLM）在“红队”安全任务中的自主攻击能力而设计的评测基准。本基准包含 70 个黑盒 CTF（夺旗赛）挑战，模拟真实 AI/ML 系统漏洞环境，要求模型独立编写 Python 代码进行漏洞发现、利用与夺旗操作，体现其计划、推理与系统操控等综合能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AIRTBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "AIRTBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 114, "released": "2025-06-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AIRTBench"}, {"aliases": [], "benchmark_id": "llm-stats-aitz-em", "caveat": "Android-In-The-Zoo (AitZ) benchmark for evaluating autonomous GUI agents on smartphones. Contains 18,643 screen-action pairs with chain-of-action-thought annotations spanning over 70 Android apps. Designed to connect perception (screen layouts and UI elements) with cognition (action decision-making) for natural language-triggered smartphone task completion.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "AITZ_EM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 115, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/aitz-em?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1969-ale-bench", "caveat": "ALE-Bench is a benchmark for evaluating AI systems on score-based algorithmic programming contests. ALE-Bench 是一个用于评估 AI 系统在基于分数的算法编程竞赛中的基准测试。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ALE-Bench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "ALE-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 116, "released": "2025-06-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ALE-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-alignbench", "caveat": "AlignBench is a comprehensive multi-dimensional benchmark for evaluating Chinese alignment of Large Language Models. It contains 8 main categories: Fundamental Language Ability, Advanced Chinese Understanding, Open-ended Questions, Writing Ability, Logical Reasoning, Mathematics, Task-oriented Role Play, and Professional Knowledge. The benchmark includes 683 real-scenario rooted queries with human-verified references and uses a rule-calibrated multi-dimensional LLM-as-Judge approach with Chain-of-Thought for evaluation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "AlignBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 117, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/alignbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1075-alignbench", "caveat": "ALIGNBENCH is a comprehensive multidimensional benchmark for evaluating LLMs’ alignment in Chinese. We tailor a humanin-the-loop data curation pipeline, containing 8 main categories, 683 real-scenario rooted queries and corresponding human verified references. AlignBench 是一个用于评估中文大语言模型对齐性能的全面、多维度的评测基准。AlignBench 构建了人类参与的数据构建流程，来保证评测数据的动态更新。AlignBench 采用多维度、规则校准的模型评价方法（LLM-as-Judge），并且结合思维链（Chain-of-Thought）生成对模型回复的多维度分析和最终的综合评分，增强了评测的高可靠性和可解释性。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AlignBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "AlignBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 118, "released": "2024-08-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AlignBench"}, {"aliases": [], "benchmark_id": "llm-stats-alpacaeval-2-0", "caveat": "AlpacaEval 2.0 is a length-controlled automatic evaluator for instruction-following language models that uses GPT-4 Turbo to assess model responses against a baseline. It evaluates models on 805 diverse instruction-following tasks including creative writing, classification, programming, and general knowledge questions. The benchmark achieves 0.98 Spearman correlation with ChatBot Arena while being fast (< 3 minutes) and affordable (< $10 in OpenAI credits). It addresses length bias in automatic evaluation through length-controlled win-rates and uses weighted scoring based on response quality.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AlpacaEval 2.0", "organization_count": 1, "organizations": ["llm_stats"], "rank": 119, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/alpacaeval-2.0?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1280-ambrosia", "caveat": "AMBROSIA is a new benchmark for recognizing and interpreting ambiguous requests in text-to-SQL. It contains questions showcasing three different types of ambiguity (scope ambiguity, attachment ambiguity, and vagueness), their interpretations, and corresponding SQL queries. AMBROSIA是识别和解释text-to-SQL中歧义请求的新基准，其中包含三种不同类型的歧义（范围歧义、附件歧义和模糊性）问题、它们的解释和相应的SQL查询。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AMBROSIA"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "AMBROSIA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 120, "released": "2024-06-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AMBROSIA"}, {"aliases": [], "benchmark_id": "llm-stats-amc-2022-23", "caveat": "American Mathematics Competition problems from the 2022-23 academic year, consisting of multiple-choice mathematics competition problems designed for high school students. These problems require advanced mathematical reasoning, problem-solving strategies, and mathematical knowledge covering topics like algebra, geometry, number theory, and combinatorics. The benchmark is derived from the official AMC competitions sponsored by the Mathematical Association of America.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "AMC_2022_23", "organization_count": 1, "organizations": ["llm_stats"], "rank": 121, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/amc-2022-23?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-amo-bench", "caveat": "AMO Bench is an olympiad-level mathematics benchmark that evaluates advanced mathematical problem-solving and multi-step reasoning on competition-style problems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "AMO Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 122, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/amo-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1966-amsbench", "caveat": "AMSbench is a benchmark of ~8000 questions to evaluate multi-modal LLMs on analog/mixed-signal circuit tasks like schematic recognition, analysis, and design. AMSbench 是一个包含约8000道题目的基准测试集，用于评估多模态大语言模型在模拟/混合信号电路任务中的表现，包括识图、分析与设计。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AMSbench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "AMSbench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 123, "released": "2025-06-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AMSbench"}, {"aliases": [], "benchmark_id": "llm-stats-android-control-high-em", "caveat": "Android device control benchmark using high exact match evaluation metric for assessing agent performance on mobile interface tasks", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Android Control High_EM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 124, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-high-em?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-android-control-low-em", "caveat": "Android control benchmark evaluating autonomous agents on mobile device interaction tasks with low exact match scoring criteria", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Android Control Low_EM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 125, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/android-control-low-em?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-androidbench", "caveat": "AndroidBench evaluates coding agents on Android application development tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "AndroidBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 126, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/androidbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-androidworld", "caveat": "AndroidWorld evaluates an agent's ability to operate in real Android GUI environments, completing multi-step tasks by perceiving screen content and executing touch/type actions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "AndroidWorld", "organization_count": 1, "organizations": ["llm_stats"], "rank": 127, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-androidworld-sr", "caveat": "AndroidWorld Success Rate (SR) benchmark - A dynamic benchmarking environment for autonomous agents operating on Android devices. Evaluates agents on 116 programmatic tasks across 20 real-world Android apps using multimodal inputs (screen screenshots, accessibility trees, and natural language instructions). Measures success rate of agents completing tasks like sending messages, creating calendar events, and navigating mobile interfaces. Published at ICLR 2025. Best current performance: 30.6% success rate (M3A agent) vs 80.0% human performance.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "AndroidWorld_SR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 128, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/androidworld-sr?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1876-aneumo", "caveat": "Based on 427 real aneurysm geometries, we synthesized 10,660 3D shapes via controlled deformation to simulate aneurysm evolution. CFD computations were performed on each shape under eight steady-state mass flow conditions, generating a total of 85,280 blood flow dynamics data covering key parameters Based on 427 real aneurysm geometries, we synthesized 10,660 3D shapes via controlled deformation to simulate aneurysm evolution. CFD computations were performed on each shape under eight steady-state mass flow conditions, generating a total of 85,280 blood flow dynamics data covering key parameters", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Aneumo"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "Aneumo", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 129, "released": "2025-05-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Aneumo"}, {"aliases": [], "benchmark_id": "llm-stats-apex", "caveat": "Apex is a challenging frontier reasoning benchmark testing advanced multi-step problem solving across difficult STEM and logical tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Apex", "organization_count": 1, "organizations": ["llm_stats"], "rank": 130, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/apex?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-apex-agents", "caveat": "APEX-Agents is a benchmark evaluating AI agents on long horizon professional tasks that require sustained reasoning, planning, and execution across complex multi-step workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "APEX-Agents", "organization_count": 1, "organizations": ["llm_stats"], "rank": 131, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-agents?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-apex-agents-aa", "caveat": "Long-horizon agentic tasks", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/apex-agents-aa"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "APEX-Agents-AA", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 132, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/apex-agents-aa"}, {"aliases": [], "benchmark_id": "llm-stats-apex-swe", "caveat": "APEX-SWE evaluates AI agents on software engineering tasks requiring multi-step coding, debugging, and verification.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "APEX-SWE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 133, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/apex-swe?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-api-bank", "caveat": "A comprehensive benchmark for tool-augmented LLMs that evaluates API planning, retrieval, and calling capabilities. Contains 314 tool-use dialogues with 753 API calls across 73 API tools, designed to assess how effectively LLMs can utilize external tools and overcome obstacles in tool leveraging.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "API-Bank", "organization_count": 1, "organizations": ["llm_stats"], "rank": 134, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/api-bank?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1093-apps", "caveat": "APPS is a benchmark for code generation. Unlike prior work in more restricted settings, our benchmark measures the ability of models to take an arbitrary natural language specification and generate satisfactory Python code. APPS 是一个代码生成评测基准，该评测基准测量模型根据任意自然语言规范生成令人满意的 Python 代码的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/APPS"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "APPS", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 135, "released": "2021-11-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/APPS"}, {"aliases": [], "benchmark_id": "opencompass-1116-aqua-rat", "caveat": "AQUA-RAT contains the algebraic word problems. The dataset consists of about 100,000 algebraic word problems with natural language rationales. AQUA-RAT 包含代数文字问题。该数据集由约 100,000 道带有自然语言推理的代数文字问题组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AQUA-RAT"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "AQUA-RAT", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 136, "released": "2017-10-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AQUA-RAT"}, {"aliases": [], "benchmark_id": "llm-stats-arc", "caveat": "The Abstraction and Reasoning Corpus (ARC) is a benchmark designed to measure human-like general fluid intelligence through grid-based reasoning tasks. It consists of 800 tasks (400 training, 400 evaluation) where each task presents input-output grids that require understanding abstract patterns and transformations. Test-takers must produce exactly correct output grids for all test inputs in a task to solve it, with 3 trials allowed per test input. ARC aims to enable fair comparisons of general intelligence between AI systems and humans using priors designed to be as close as possible to innate human priors.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Arc", "organization_count": 1, "organizations": ["llm_stats"], "rank": 137, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-arc-agi", "caveat": "The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) is a benchmark designed to test general intelligence and abstract reasoning capabilities through visual grid-based transformation tasks. Each task consists of 2-5 demonstration pairs showing input grids transformed into output grids according to underlying rules, with test-takers required to infer these rules and apply them to novel test inputs. The benchmark uses colored grids (up to 30x30) with 10 discrete colors/symbols, designed to measure human-like general fluid intelligence and skill-acquisition efficiency with minimal prior knowledge.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ARC-AGI", "organization_count": 1, "organizations": ["llm_stats"], "rank": 138, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-arc-agi-v2", "caveat": "ARC-AGI-2 is an upgraded benchmark for measuring abstract reasoning and problem-solving abilities in AI systems through visual grid transformation tasks. It evaluates fluid intelligence via input-output grid pairs (1x1 to 30x30) using colored cells (0-9), requiring models to identify underlying transformation rules from demonstration examples and apply them to test cases. Designed to be easy for humans but challenging for AI, focusing on core cognitive abilities like spatial reasoning, pattern recognition, and compositional generalization.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ARC-AGI v2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 139, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-v2?top_n=500"}, {"aliases": ["ARC-AGI-3", "ARC-AGI 3"], "benchmark_id": "arc_agi_3", "caveat": "Interactive multi-step format, so the agent harness is part of the measurement. Scores are low and spread wide, which makes small absolute gaps look larger than they are.", "document_count": 1, "document_ids": ["model_reports:anthropic_claude_opus_5_system_card"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ARC-AGI-3", "organization_count": 1, "organizations": ["Anthropic"], "rank": 140, "released": "2026-01-15", "source": "model_reports", "url": "https://arcprize.org/arc-agi/3/"}, {"aliases": [], "benchmark_id": "llm-stats-arc-agi-3", "caveat": "ARC-AGI-3 is the third-generation Abstraction and Reasoning Corpus benchmark, an interactive-reasoning evaluation designed to measure fluid, novel problem-solving ability that remains far from saturated for frontier models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ARC-AGI-3", "organization_count": 1, "organizations": ["llm_stats"], "rank": 141, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-agi-3?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-arc-c", "caveat": "The AI2 Reasoning Challenge (ARC) Challenge Set is a multiple-choice question-answering benchmark containing grade-school level science questions that require advanced reasoning capabilities. ARC-C specifically contains questions that were answered incorrectly by both retrieval-based and word co-occurrence algorithms, making it a particularly challenging subset designed to test commonsense reasoning abilities in AI systems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ARC-C", "organization_count": 1, "organizations": ["llm_stats"], "rank": 142, "released": "2018-03-14", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-c?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-502-arc-c", "caveat": "The AI2’s Reasoning Challenge (ARC) dataset is a multiple-choice question-answering dataset, containing questions from science exams from grade 3 to grade 9. The dataset is split in two partitions: Easy and Challenge, where the latter partition contains the more difficult questions that require reasoning. Most of the questions have 4 answer choices, with <1% of all the questions having either 3 or 5 answer choices. AI2的推理挑战（ARC）数据集是一个多项选择问题回答数据集，包含了从三年级到九年级的科学考试中提取的问题。该数据集分为两个部分：简单和挑战，其中后者包含了需要推理能力的更难的问题。大多数问题有4个答案选项，仅有不到1％的问题有3个或5个答案选项。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ARC-c"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "ARC-c", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 143, "released": "2018-03-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ARC-c"}, {"aliases": [], "benchmark_id": "llm-stats-arc-e", "caveat": "ARC-E (AI2 Reasoning Challenge - Easy Set) is a subset of grade-school level, multiple-choice science questions that requires knowledge and reasoning capabilities. Part of the AI2 Reasoning Challenge dataset containing 5,197 questions that test scientific reasoning and factual knowledge. The Easy Set contains questions that are answerable by retrieval-based and word co-occurrence algorithms, making them more accessible than the Challenge Set.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ARC-E", "organization_count": 1, "organizations": ["llm_stats"], "rank": 144, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arc-e?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-503-arc-e", "caveat": "The AI2’s Reasoning Challenge (ARC) dataset is a multiple-choice question-answering dataset, containing questions from science exams from grade 3 to grade 9. The dataset is split in two partitions: Easy and Challenge, where the latter partition contains the more difficult questions that require reasoning. Most of the questions have 4 answer choices, with <1% of all the questions having either 3 or 5 answer choices. AI2的推理挑战（ARC）数据集是一个多项选择问题回答数据集，包含了从三年级到九年级的科学考试中提取的问题。该数据集分为两个部分：简单和挑战，其中后者包含了需要推理能力的更难的问题。大多数问题有4个答案选项，仅有不到1％的问题有3个或5个答案选项。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ARC-e"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "ARC-e", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 145, "released": "2018-03-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ARC-e"}, {"aliases": [], "benchmark_id": "llm-stats-arcagi2", "caveat": "ARC-AGI-2 is the second-generation Abstraction and Reasoning Corpus benchmark measuring fluid, general reasoning and abstraction.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ArcAGI2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 146, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arcagi2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-arena-hard", "caveat": "Arena-Hard-Auto is an automatic evaluation benchmark for instruction-tuned LLMs consisting of 500 challenging real-world prompts curated by BenchBuilder. It includes open-ended software engineering problems, mathematical questions, and creative writing tasks. The benchmark uses LLM-as-a-Judge methodology with GPT-4.1 and Gemini-2.5 as automatic judges to approximate human preference. Arena-Hard achieves 98.6% correlation with human preference rankings and provides 3x higher separation of model performances compared to MT-Bench, making it highly effective for distinguishing between models of similar quality.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Arena Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 147, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-arena-hard-v2", "caveat": "Arena-Hard-Auto v2 is a challenging benchmark consisting of 500 carefully curated prompts sourced from Chatbot Arena and WildChat-1M, designed to evaluate large language models on real-world user queries. The benchmark covers diverse domains including open-ended software engineering problems, mathematics, creative writing, and technical problem-solving. It uses LLM-as-a-Judge for automatic evaluation, achieving 98.6% correlation with human preference rankings while providing 3x higher separation of model performances compared to MT-Bench. The benchmark emphasizes prompt specificity, complexity, and domain knowledge to better distinguish between model capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Arena-Hard v2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 148, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arena-hard-v2?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2075-arena-hard-auto", "caveat": "Arena-Hard-Auto is an automated benchmark for instruction-tuned LLMs, designed to efficiently approximate human preferences. Arena-Hard-Auto 是一个用于评估 LLM 的基准，自动甄选 500 条高难度开放式提示，从模型区分度、人类偏好一致性与提示质量三维度进行严苛评测。依托 BenchBuilder 管道、主题建模与 LLM 裁判，实现众包数据→筛选→评分的全自动闭环。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Arena-Hard-Auto"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "Arena-Hard-Auto", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 149, "released": "2024-04-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Arena-Hard-Auto"}, {"aliases": [], "benchmark_id": "opencompass-2370-argusinspection", "caveat": "As Multimodal Large Language Models (MLLMs) continue to evolve, their cognitive and reasoning capabilities have seen remarkable progress. However, challenges in visual fine-grained perception and commonsense causal inference persist. This paper introduces Argus Inspection, a multimodal benchmark wit As Multimodal Large Language Models (MLLMs) continue to evolve, their cognitive and reasoning capabilities have seen remarkable progress. However, challenges in visual fine-grained perception and commonsense causal inference persist. This paper introduces Argus Inspection, a multimodal benchmark wit", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ArgusInspection"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ArgusInspection", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 150, "released": "2025-10-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ArgusInspection"}, {"aliases": [], "benchmark_id": "llm-stats-arkitscenes", "caveat": "ARKitScenes evaluates 3D scene understanding and spatial reasoning in AR/VR contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "ARKitScenes", "organization_count": 1, "organizations": ["llm_stats"], "rank": 151, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arkitscenes?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-artifacts-bench", "caveat": "Artifacts Bench evaluates a model's ability to generate visual code artifacts, measuring the quality of generated interactive and visual front-end outputs from natural-language requests.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "frontend_development", "name": "Artifacts Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 152, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/artifacts-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2052-artifactsbench", "caveat": "ArtifactsBench is a benchmark with 1,825 tasks for evaluating LLM-generated visual and interactive code. It addresses the gap of traditional benchmarks that focus only on algorithmic correctness by assessing visual fidelity and user interaction. rtifactsBench 是一个专注于弥合传统代码评测中“视觉-交互”鸿沟的新型基准。它旨在全面评估大语言模型（LLM）生成动态可视化与交互式代码的能力，而非仅仅考核算法正确性。\n该基准包含1825个真实世界的任务，并开创了一套自动化多模态评估流程：系统会自动渲染代码、捕捉其视觉与交互行为，再由一个多模态大模型（MLLM）依据详细清单进行评分。该流程与人类专家判断的一致性高达94.4%，证明了其高度可靠性。\nArtifactsBench 已将数据集、评估框架及工具链完全开源，为社区提供了一个可扩展且精准的工具，以推动能创造丰富用户体验的新一代生成模型的发展。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ArtifactsBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ArtifactsBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 153, "released": "2025-07-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ArtifactsBench"}, {"aliases": [], "benchmark_id": "llm-stats-artificial-analysis", "caveat": "Artificial Analysis benchmark evaluates AI models across quality, speed, and pricing dimensions, providing a composite assessment of model capabilities for real-world usage.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "Artificial Analysis", "organization_count": 1, "organizations": ["llm_stats"], "rank": 154, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/artificial-analysis?top_n=500"}, {"aliases": [], "benchmark_id": "arxivmath", "caveat": "Competition-style mathematics drawn from arXiv; test-time compute budget is part of the result.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "math", "name": "ArXivMath", "organization_count": 1, "organizations": ["Tencent"], "rank": 155, "released": null, "source": "model_reports", "url": "https://matharena.ai/arxivmath"}, {"aliases": [], "benchmark_id": "llm-stats-arxivmath", "caveat": "ArXivMath is a final-answer benchmark of research-level mathematics maintained by MathArena. Problems are extracted monthly from recent arXiv paper abstracts, then filtered through automated and manual checks to ensure they are self-contained, non-trivial, and verifiable. Because problems are drawn from active research, the benchmark is more realistic and more closely connected to mathematical research than contest or olympiad benchmarks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "ArXivMath", "organization_count": 1, "organizations": ["llm_stats"], "rank": 156, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/arxivmath?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1114-asdiv", "caveat": "ASDiv is a new MWP corpus that contains diverse lexicon patterns with wide problem type coverage. Each problem provides consistent equations and answers. It is further annotated with the corresponding problem type and grade level. ASDiv 是一个新的数学文字问题（MWP）语料库，包含多样的词汇模式，覆盖广泛的问题类型。每个问题提供对应的方程和答案。它进一步标注了相应的问题类型和年级水平，可用于测试系统的能力，并指明问题的难度等级。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ASDiv"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "ASDiv", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 157, "released": "2020-07-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ASDiv"}, {"aliases": [], "benchmark_id": "opencompass-1991-assetopsbench", "caveat": "AssetOpsBench is a comprehensive benchmark designed to evaluate the performance of large language models (LLMs) and AI agents in complex asset operation and maintenance tasks. AssetOpsBench 是一个专注于评估大语言模型（LLM）和智能体在资产运维领域复杂任务中实际表现的多维度评测基准。该基准旨在检验模型在工业场景下的任务规划、多步推理、工具调用、安全合规性以及领域知识理解等核心能力，覆盖设备维护、异常诊断、风险评估等典型运维场景。测试集包含 1,000 个高质量样本，涉及 5 大类任务和 20 余种细分领域，数据来源于真实运维手册、工单记录及专家验证案例。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AssetOpsBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "AssetOpsBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 158, "released": "2025-06-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AssetOpsBench"}, {"aliases": [], "benchmark_id": "llm-stats-community-ed90e889-4678-4fbd-98ab-0e654f4bf35e", "caveat": null, "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3Aed90e889-4678-4fbd-98ab-0e654f4bf35e?top_n=500"], "document_share": 0.0008271298593879239, "domain": "other", "name": "atlas", "organization_count": 1, "organizations": ["llm_stats"], "rank": 159, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Aed90e889-4678-4fbd-98ab-0e654f4bf35e?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-attaq", "caveat": "AttaQ is a unique dataset containing adversarial examples in the form of questions designed to provoke harmful or inappropriate responses from large language models. The benchmark evaluates safety vulnerabilities by using specialized clustering techniques that analyze both the semantic similarity of input attacks and the harmfulness of model responses, facilitating targeted improvements to model safety mechanisms.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "AttaQ", "organization_count": 1, "organizations": ["llm_stats"], "rank": 160, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/attaq?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1835-audiojailbreak", "caveat": "LAMs face jailbreak risks. AJailBench, our new benchmark, reveals leading LAMs lack robustness. Subtle audio perturbations significantly degrade their safety. We release AJailBench for research. LAM 面临越狱风险。我们新的基准测试 AJailBench 揭示，领先的 LAM 缺乏稳健性。细微的音频干扰会显著降低其安全性。我们发布 AJailBench 进行研究。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AudioJailbreak"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "AudioJailbreak", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 161, "released": "2025-05-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AudioJailbreak"}, {"aliases": [], "benchmark_id": "opencompass-1908-audiotrust", "caveat": "AudioTrust is a comprehensive trust evaluation framework for Audio Large Language Models (ALLMs) that effectively reveals potential risks in six dimensions: fairness, hallucination, security, privacy, robustness, and authentication. It aggregates over 4,420 real-world audio/text data samples, coveri AudioTrust针对Audio Large Language Models（ALLMs）的全方位可信评估框架，有效揭示音频大模型在公平性、幻觉、安全、隐私、鲁棒性和身份验证六大维度的潜在风险。汇集4,420+条真实场景音频/文本数据，覆盖日常对话、紧急呼叫、语音助手等18种实验设置，设计9项音频特定评测指标，构建自动化评估流水线。主要发现：闭源模型在鲁棒性和安全防护上表现更佳，开源模型对隐私和公平性仍存盲区；多数ALLMs对性别、口音、年龄等敏感属性存在系统性偏见。期待研究者基于AudioTrust继续优化音频大模型，共同推动更安全、可信的AI音频生态发展！", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AudioTrust"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "AudioTrust", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 162, "released": "2025-06-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AudioTrust"}, {"aliases": [], "benchmark_id": "opencompass-2078-autoadvexbench", "caveat": "AutoAdvExBench is a benchmark designed to evaluate large language models' (LLMs) ability to autonomously exploit adversarial example defenses, directly measuring LLMs' success on tasks regularly performed by machine learning security experts. AutoAdvExBench 是一个评估大型语言模型（LLMs）自主利用对抗性样本防御能力的基准，直接衡量LLMs在机器学习安全专家任务上的成功率。它主要评估模型理解学术论文、代码实现及生成对抗性攻击的能力。测试集包含75个对抗性样本防御实现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AutoAdvExBench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "AutoAdvExBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 163, "released": "2025-03-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AutoAdvExBench"}, {"aliases": [], "benchmark_id": "llm-stats-autologi", "caveat": "AutoLogi is an automated method for synthesizing open-ended logic puzzles to evaluate reasoning abilities of Large Language Models. The benchmark addresses limitations of existing multiple-choice reasoning evaluations by featuring program-based verification and controllable difficulty levels. It includes 1,575 English and 883 Chinese puzzles, enabling more reliable evaluation that better distinguishes models' reasoning capabilities across languages.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AutoLogi", "organization_count": 1, "organizations": ["llm_stats"], "rank": 164, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/autologi?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-automationbench", "caveat": "AutomationBench is a tool-use benchmark that evaluates AI agents on automating real-world workflows, testing their ability to orchestrate tools and complete multi-step automation tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AutomationBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 165, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-automationbench-aa", "caveat": "Agentic SaaS workflows", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/automationbench-aa"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "AutomationBench-AA", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 166, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/automationbench-aa"}, {"aliases": [], "benchmark_id": "llm-stats-automationbench-aa", "caveat": "AutomationBench-AA is Artificial Analysis's independently run version of AutomationBench, covering 657 real-world SaaS workflow tasks across 40 simulated applications (e.g. Gmail, Slack, Salesforce, HubSpot). It scores the share of objectives an agent completes without violating business guardrails, using a private held-out task set.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "AutomationBench-AA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 167, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/automationbench-aa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1396-av-odyssey-bench", "caveat": "AV-Odyssey Bench. This benchmark encompasses 26 different tasks and 4,555 carefully crafted problems, each incorporating text, visual, and audio components. All data are newly collected and annotated by humans, not from any existing audio-visual dataset. AV-Odyssey Bench. This benchmark encompasses 26 different tasks and 4,555 carefully crafted problems, each incorporating text, visual, and audio components. All data are newly collected and annotated by humans, not from any existing audio-visual dataset.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AV-Odyssey-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "AV-Odyssey-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 168, "released": "2024-12-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AV-Odyssey-Bench"}, {"aliases": [], "benchmark_id": "opencompass-526-ax-b", "caveat": "AX-b is a broad-coverage diagnostic task, which requires to determine the logical relation between the given sentence pair, with three relations: entailment, contradiction and neutral. This task is selected from a subset of the GLUE broad-coverage diagnostic dataset, mainly used to test the model's understanding ability in grammar, semantics, world knowledge and so on. AX-b是一个广覆盖诊断任务，要求根据给定的句子对，判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。这个任务是从GLUE的广覆盖诊断数据集中选取了一部分数据，主要用来测试模型在语法、语义、世界知识等方面的理解能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AX-b"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "AX-b", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 169, "released": "2019-05-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AX-b"}, {"aliases": [], "benchmark_id": "opencompass-527-ax-g", "caveat": "AX-g is a Winogender diagnostic task, which requires to determine which noun the pronoun refers to according to the given sentence and pronoun. This task is selected from a subset of the Winogender dataset, mainly used to test the model's ability in dealing with gender bias and discrimination. AX-g是一个Winogender诊断任务，要求根据给定的句子和代词，判断代词指代的是哪个名词。这个任务是从Winogender数据集中选取了一部分数据，主要用来测试模型在处理性别偏见和性别歧视方面的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AX-g"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "AX-g", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 170, "released": "2019-07-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AX-g"}, {"aliases": [], "benchmark_id": "opencompass-2084-axbench", "caveat": "AXBENCH is a benchmark for large-scale evaluation of Language Model (LLM) control methods using synthetic data, focusing on fine-grained steering for safety and reliability. AXBENCH 是一个旨在评估LLM控制能力的基准。它通过概念检测和模型操控（包含概念、指令、流畅度）评估，旨在实现安全可靠的细粒度操控。基准使用大规模合成数据集，并集成了多种基线方法。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/AXBENCH"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "AXBENCH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 171, "released": "2025-01-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/AXBENCH"}, {"aliases": [], "benchmark_id": "opencompass-1266-babilong", "caveat": "BABILong is designed to test language models' ability to reason across facts distributed in extremely long documents. It contains a diverse set of 20 reasoning tasks, including fact chaining, simple induction, deduction, counting, and handling lists/sets. BABILong旨在测试语言模型对分布在极长文档中的事实进行推理的能力，涵盖事实链接、简单归纳、推导、计数和处理列表/集合等20种各类推理任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BABILong"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "BABILong", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 172, "released": "2024-06-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BABILong"}, {"aliases": ["BabyVision"], "benchmark_id": "babyvision", "caveat": "Multimodal vision benchmark. Temperature=1.0, top_p=0.95, max context 164K tokens. Images resized to shorter side at least 1.5K pixels.", "document_count": 1, "document_ids": ["model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "BabyVision", "organization_count": 1, "organizations": ["Z.ai"], "rank": 173, "released": "2025-12-01", "source": "model_reports", "url": "https://github.com/babyvision/babyvision"}, {"aliases": [], "benchmark_id": "llm-stats-babyvision", "caveat": "A benchmark for early-stage visual reasoning and perception on child-like vision tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "BabyVision", "organization_count": 1, "organizations": ["llm_stats"], "rank": 174, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/babyvision?top_n=500"}, {"aliases": ["BankerToolBench", "Banker Tool Bench", "BTB"], "benchmark_id": "bankertoolbench", "caveat": "End-to-end investment-banking tasks produce spreadsheets, presentations, and documents; the agent harness, financial-data tools, and rubric-grader configuration are part of the score.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "professional", "name": "BankerToolBench", "organization_count": 1, "organizations": ["Tencent"], "rank": 175, "released": null, "source": "model_reports", "url": "https://github.com/Handshake-AI-Research/bankertoolbench"}, {"aliases": [], "benchmark_id": "llm-stats-bankertoolbench", "caveat": "BankerToolBench is a public benchmark that evaluates models on banking and finance tool-use tasks. Models are scored against dataset rubrics, measuring their ability to correctly invoke tools and complete multi-step financial workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "finance", "name": "BankerToolBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 176, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bankertoolbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bbh", "caveat": "Big-Bench Hard (BBH) is a suite of 23 challenging tasks selected from BIG-Bench for which prior language model evaluations did not outperform the average human-rater. These tasks require multi-step reasoning across diverse domains including arithmetic, logical reasoning, reading comprehension, and commonsense reasoning. The benchmark was designed to test capabilities believed to be beyond current language models and focuses on evaluating complex reasoning skills including temporal understanding, spatial reasoning, causal understanding, and deductive logical reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "BBH", "organization_count": 1, "organizations": ["llm_stats"], "rank": 177, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bbh?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-539-bbh", "caveat": "BIG-Bench Hard (BBH) is a subset of the BIG-Bench, a diverse evaluation suite for language models. BBH focuses on a suite of 23 challenging tasks from BIG-Bench that were found to be beyond the capabilities of current language models. BIG Bench-Hard（BBH）是BIG Bench的一个子集，它是一个用于语言模型的多样化评估套件。BBH专注于BIG Bench的23项具有挑战性的任务，这些任务被发现超出了当前语言模型的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BBH"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "BBH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 178, "released": "2022-10-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BBH"}, {"aliases": [], "benchmark_id": "llm-stats-bc-vl", "caveat": "BC-VL is a vision-language benchmark for knowledge-grounded multimodal question answering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "BC-VL", "organization_count": 1, "organizations": ["llm_stats"], "rank": 179, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bc-vl?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-beam-128k", "caveat": "Beam 128K evaluates reasoning over long inputs at a 128K-token context length.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "Beam 128K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 180, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/beam-128k?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1086-belebele", "caveat": "BELEBELE, a multiple-choice machine reading comprehension (MRC) dataset spanning 122 language variants. Significantly expanding the language coverage of natural language understanding (NLU) benchmarks, this dataset enables the evaluation of text models in\nhigh-, medium-, and low-resource languages. BELEBELE 是一个多项选择机器阅读理解（MRC）数据集，涵盖 122 种语言变体。该数据集显著扩展了自然语言理解（NLU）基准的语言覆盖范围，使得可以在高、中、低资源语言中评估文本模型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Belebele"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "Belebele", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 181, "released": "2024-07-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Belebele"}, {"aliases": [], "benchmark_id": "llm-stats-benchcad", "caveat": "BenchCAD is a benchmark for programmatic CAD reasoning built from 17,900 execution-verified CadQuery programs spanning 106 industrial part families, roughly half anchored to real ISO, DIN, EN, ASME, and IEC specification tables. It decomposes CAD capability into matched tasks; the Vision2Code task requires models to generate CadQuery code from multi-view renders, scored by voxel IoU against the reference geometry.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "BenchCAD", "organization_count": 1, "organizations": ["llm_stats"], "rank": 182, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-benchcad-with-python-tool", "caveat": "BenchCAD variant evaluated with access to a Python tool for programmatic CAD reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "BenchCAD (with Python tool)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 183, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/benchcad-with-python-tool?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1562-benchmax", "caveat": "BenchMAX is a comprehensive, high-quality, and multiway parallel multilingual benchmark comprising 10 tasks designed to assess crucial capabilities across 17 diverse language. BenchMAX 是一个全面、高质量的多向并行多语言基准，包含 10 个任务，旨在评估 17 种不同语言的关键能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BenchMAX"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "BenchMAX", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 184, "released": "2025-02-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BenchMAX"}, {"aliases": [], "benchmark_id": "llm-stats-beyond-aime", "caveat": "Beyond AIME is a difficult mathematical reasoning benchmark designed to test deeper reasoning chains and harder decomposition than standard AIME-style problem sets.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "Beyond AIME", "organization_count": 1, "organizations": ["llm_stats"], "rank": 185, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/beyond-aime?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bfcl", "caveat": "The Berkeley Function Calling Leaderboard (BFCL) is the first comprehensive and executable function call evaluation dedicated to assessing Large Language Models' ability to invoke functions. It evaluates serial and parallel function calls across multiple programming languages (Python, Java, JavaScript, REST API) using a novel Abstract Syntax Tree (AST) evaluation method. The benchmark consists of over 2,000 question-function-answer pairs covering diverse application domains and complex use cases including multiple function calls, parallel function calls, and multi-turn interactions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BFCL", "organization_count": 1, "organizations": ["llm_stats"], "rank": 186, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bfcl-v2", "caveat": "Berkeley Function Calling Leaderboard (BFCL) v2 is a comprehensive benchmark for evaluating large language models' function calling capabilities. It features 2,251 question-function-answer pairs with enterprise and OSS-contributed functions, addressing data contamination and bias through live, user-contributed scenarios. The benchmark evaluates AST accuracy, executable accuracy, irrelevance detection, and relevance detection across multiple programming languages (Python, Java, JavaScript) and includes complex real-world function calling scenarios with multi-lingual prompts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BFCL v2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 187, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bfcl-v3", "caveat": "Berkeley Function Calling Leaderboard v3 (BFCL-v3) is an advanced benchmark that evaluates large language models' function calling capabilities through multi-turn and multi-step interactions. It introduces extended conversational exchanges where models must retain contextual information across turns and execute multiple internal function calls for complex user requests. The benchmark includes 1000 test cases across domains like vehicle control, trading bots, travel booking, and file system management, using state-based evaluation to verify both system state changes and execution path correctness.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BFCL-v3", "organization_count": 1, "organizations": ["llm_stats"], "rank": 188, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bfcl-v4", "caveat": "Berkeley Function Calling Leaderboard V4 (BFCL-V4) evaluates LLMs on their ability to accurately call functions and APIs, including simple, multiple, parallel, and nested function calls across diverse programming scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "BFCL-V4", "organization_count": 1, "organizations": ["llm_stats"], "rank": 189, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v4?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bfcl-v3-multiturn", "caveat": "Berkeley Function Calling Leaderboard (BFCL) V3 MultiTurn benchmark that evaluates large language models' ability to handle multi-turn and multi-step function calling scenarios. The benchmark introduces complex interactions requiring models to manage sequential function calls, handle conversational context across multiple turns, and make dynamic decisions about when and how to use available functions. BFCL V3 uses state-based evaluation by verifying the actual state of API systems after function execution, providing more realistic assessment of function calling capabilities in agentic applications.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BFCL_v3_MultiTurn", "organization_count": 1, "organizations": ["llm_stats"], "rank": 190, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bfcl-v3-multiturn?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-big-bench-audio", "caveat": "Big Bench Audio is an audio reasoning benchmark adapted from a subset of Big Bench Hard, with text questions converted to spoken audio. It evaluates the reasoning ability of speech-to-speech and audio language models on tasks delivered as audio input, with accuracy scored by an independent evaluation (Artificial Analysis).", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Big Bench Audio", "organization_count": 1, "organizations": ["llm_stats"], "rank": 191, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-audio?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-big-finance-bench", "caveat": "Big Finance Bench evaluates models on complex financial-analysis tasks that require retrieving and reasoning over financial documents and performing multi-step quantitative work.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Big Finance Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 192, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-finance-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-big-bench", "caveat": "Beyond the Imitation Game Benchmark (BIG-bench) is a collaborative benchmark consisting of 204+ tasks designed to probe large language models and extrapolate their future capabilities. It covers diverse domains including linguistics, mathematics, common-sense reasoning, biology, physics, social bias, software development, and more. The benchmark focuses on tasks believed to be beyond current language model capabilities and includes both English and non-English tasks across multiple languages.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "BIG-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 193, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench?top_n=500"}, {"aliases": ["BigBench Extra Hard", "BBEH", "BIG-Bench Extra Hard"], "benchmark_id": "bigbench_extra_hard", "caveat": "Successor to BBH after saturation; per-task variance is high.", "document_count": 1, "document_ids": ["model_reports:google_gemma_4_model_card"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BIG-Bench Extra Hard", "organization_count": 1, "organizations": ["Google"], "rank": 194, "released": "2025-02-26", "source": "model_reports", "url": "https://arxiv.org/abs/2502.19187"}, {"aliases": [], "benchmark_id": "llm-stats-big-bench-extra-hard", "caveat": "BIG-Bench Extra Hard (BBEH) is a challenging benchmark that replaces each task in BIG-Bench Hard with a novel task that probes similar reasoning capabilities but exhibits significantly increased difficulty. The benchmark contains 23 tasks testing diverse reasoning skills including many-hop reasoning, causal understanding, spatial reasoning, temporal arithmetic, geometric reasoning, linguistic reasoning, logic puzzles, and humor understanding. Designed to address saturation on existing benchmarks where state-of-the-art models achieve near-perfect scores, BBEH shows substantial room for improvement with best models achieving only 9.8-44.8% average accuracy.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BIG-Bench Extra Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 195, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-extra-hard?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-big-bench-hard", "caveat": "BIG-Bench Hard (BBH) is a subset of 23 challenging BIG-Bench tasks selected because prior language model evaluations did not outperform average human-rater performance. The benchmark contains 6,511 evaluation examples testing various forms of multi-step reasoning including arithmetic, logical reasoning (Boolean expressions, logical deduction), geometric reasoning, temporal reasoning, and language understanding. Tasks require capabilities such as causal judgment, object counting, navigation, pattern recognition, and complex problem solving.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "BIG-Bench Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 196, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/big-bench-hard?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bigcodebench", "caveat": "A benchmark that challenges LLMs to invoke multiple function calls as tools from 139 libraries and 7 domains for 1,140 fine-grained programming tasks. Evaluates code generation with diverse function calls and complex instructions, featuring two variants: Complete (code completion based on comprehensive docstrings) and Instruct (generating code from natural language instructions).", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BigCodeBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 197, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1253-bigcodebench", "caveat": "BigCodeBench is a benchmark that challenges LLMs to invoke multiple function calls as tools from 139 libraries and 7 domains for 1,140 fine-grained tasks. BigCodeBench用于评估LLM的代码生成能力，包含1140个可以调用139个库和7个域的多个函数来完成的细粒度任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BigCodeBench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "BigCodeBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 198, "released": "2024-06-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BigCodeBench"}, {"aliases": [], "benchmark_id": "llm-stats-bigcodebench-full", "caveat": "A comprehensive benchmark that evaluates large language models' ability to solve complex, practical programming tasks via code generation. Contains 1,140 fine-grained tasks across 7 domains using function calls from 139 libraries. Challenges LLMs to invoke multiple function calls as tools and handle complex instructions for realistic software engineering and general-purpose reasoning tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BigCodeBench-Full", "organization_count": 1, "organizations": ["llm_stats"], "rank": 199, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-full?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bigcodebench-hard", "caveat": "BigCodeBench-Hard is a subset of 148 challenging programming tasks from BigCodeBench, designed to evaluate large language models' ability to solve complex, real-world programming problems. These tasks require diverse function calls from multiple libraries across 7 domains including computation, networking, data analysis, and visualization. The benchmark tests compositional reasoning and the ability to implement complex instructions that span 139 libraries with an average of 2.8 libraries per task.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BigCodeBench-Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 200, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bigcodebench-hard?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1680-bigobench", "caveat": "BigO(Bench)是一个包含约 300 个需要用 Python 解决的代码问题的基准测试，以及 3,105 个编码问题和 1,190,250 个解决方案用于训练，以评估LLMs能否找到代码解决方案的时间-空间复杂度，或者生成符合时间-空间复杂度要求的代码解决方案。 BigO(Bench)是一个包含约 300 个需要用 Python 解决的代码问题的基准测试，以及 3,105 个编码问题和 1,190,250 个解决方案用于训练，以评估LLMs能否找到代码解决方案的时间-空间复杂度，或者生成符合时间-空间复杂度要求的代码解决方案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BigOBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "BigOBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 201, "released": "2025-03-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BigOBench"}, {"aliases": [], "benchmark_id": "llm-stats-biolp-bench", "caveat": "BioLP-Bench is a model-graded evaluation measuring ability to find and correct mistakes in common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "BioLP-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 202, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/biolp-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-biomysterybench", "caveat": "BioMysteryBench evaluates a model's ability to reason through challenging molecular biology problems, reporting performance on a hard subset and on the subset of problems solved by human experts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BioMysteryBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 203, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/biomysterybench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bird-sql-dev", "caveat": "BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQLs) is a comprehensive text-to-SQL benchmark containing 12,751 question-SQL pairs across 95 databases (33.4 GB total) spanning 37+ professional domains. It evaluates large language models' ability to convert natural language to executable SQL queries in real-world scenarios with complex database schemas and dirty data.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Bird-SQL (dev)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 204, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bird-sql-%28dev%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-bixbench", "caveat": "BixBench is a benchmark for real-world bioinformatics and computational biology data analysis. It evaluates AI models on multi-step scientific workflows that require code execution, statistical reasoning, and biological domain knowledge to interpret experimental data.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BixBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 205, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/bixbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-blink", "caveat": "BLINK: Multimodal Large Language Models Can See but Not Perceive. A benchmark for multimodal language models focusing on core visual perception abilities. Reformats 14 classic computer vision tasks into 3,807 multiple-choice questions paired with single or multiple images and visual prompting. Tasks include relative depth estimation, visual correspondence, forensics detection, multi-view reasoning, counting, object localization, and spatial reasoning that humans can solve 'within a blink'.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "BLINK", "organization_count": 1, "organizations": ["llm_stats"], "rank": 206, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/blink?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1365-blink", "caveat": "BLINK focuses on MLLMs' core visual perception abilities. It contains 3,807 multiple-choice questions spanning 14 classic computer vision tasks. BLINK用于评估多模态大模型的视觉感知能力，包含来自14个经典计算机视觉任务的3807道多项选择题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BLINK"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "BLINK", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 207, "released": "2024-04-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BLINK"}, {"aliases": ["Blueprint-Bench 2", "Blueprint-Bench", "BlueprintBench 2"], "benchmark_id": "blueprint_bench_2", "caveat": "Spatial-reasoning set: 50 apartments, ~20 photos each, scored by a connectivity-graph grader, with a public leaderboard on the Andon Labs eval page (run in-house; dataset not openly downloadable). Builds on the original Blueprint-Bench paper (arXiv:2509.25229), a distinct instrument released 2025-09-24 whose scores must not be compared across versions.", "document_count": 1, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "Blueprint-Bench 2", "organization_count": 1, "organizations": ["Anthropic"], "rank": 208, "released": "2026-05-04", "source": "model_reports", "url": "https://andonlabs.com/evals/blueprint-bench-2"}, {"aliases": [], "benchmark_id": "llm-stats-blueprint-bench-2", "caveat": "Blueprint-Bench 2 is an agentic spatial reasoning benchmark that evaluates a model's ability to understand, plan, and reason over architectural blueprints and other structured spatial documents. Scores are reported as a normalized score.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Blueprint-Bench 2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 209, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/blueprint-bench-2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-boolq", "caveat": "BoolQ is a reading comprehension dataset for yes/no questions containing 15,942 naturally occurring examples. Each example consists of a question, passage, and boolean answer, where questions are generated in unprompted and unconstrained settings. The dataset challenges models with complex, non-factoid information requiring entailment-like inference to solve.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BoolQ", "organization_count": 1, "organizations": ["llm_stats"], "rank": 210, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/boolq?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-510-boolq", "caveat": "BoolQ is a question answering dataset for yes/no questions containing 15942 examples. These questions are naturally occurring ---they are generated in unprompted and unconstrained settings. Each example is a triplet of (question, passage, answer), with the title of the page as optional additional context. BoolQ是一个包含15942个示例的是/否问题的问答数据集。这些问题是自然生成的——即在无prompt和无约束的环境中产生的。每个例子都是一个三元组(问题、段落、答案)，页面标题是可选的附加上下文。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BoolQ"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "BoolQ", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 211, "released": "2019-05-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BoolQ"}, {"aliases": [], "benchmark_id": "opencompass-1571-bright", "caveat": "BRIGHT is the first text retrieval benchmark that requires intensive reasoning to retrieve relevant documents. BRIGHT 是第一个需要大量推理来检索相关文档的文本检索基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BRIGHT"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "BRIGHT", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 212, "released": "2024-10-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BRIGHT"}, {"aliases": [], "benchmark_id": "brokenarxiv", "caveat": "arXiv proofs with planted errors; tests whether a model catches a broken argument instead of reproducing it.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "math", "name": "BrokenArXiv", "organization_count": 1, "organizations": ["Tencent"], "rank": 213, "released": null, "source": "model_reports", "url": "https://matharena.ai/brokenarxiv"}, {"aliases": [], "benchmark_id": "llm-stats-browsecomp", "caveat": "BrowseComp is a benchmark comprising 1,266 questions that challenge AI agents to persistently navigate the internet in search of hard-to-find, entangled information. The benchmark measures agents' ability to exercise persistence in information gathering, demonstrate creativity in web navigation, and find concise, verifiable answers. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BrowseComp", "organization_count": 1, "organizations": ["llm_stats"], "rank": 214, "released": "2025-04-10", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-browsecomp-long-128k", "caveat": "A challenging benchmark for evaluating web browsing agents' ability to persistently navigate the internet and find hard-to-locate, entangled information. Comprises 1,266 questions requiring strategic reasoning, creative search, and interpretation of retrieved content, with short and easily verifiable answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BrowseComp Long Context 128k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 215, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-browsecomp-long-256k", "caveat": "BrowseComp is a benchmark for measuring the ability of agents to browse the web, comprising 1,266 questions that require persistently navigating the internet in search of hard-to-find, entangled information. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers. The benchmark focuses on questions where answers are obscure, time-invariant, and well-supported by evidence scattered across the open web.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BrowseComp Long Context 256k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 216, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-long-256k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-browsecomp-vl", "caveat": "BrowseComp-VL is the vision-language variant of BrowseComp, evaluating multimodal models on web browsing comprehension tasks that require processing visual web page content alongside text.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "BrowseComp-VL", "organization_count": 1, "organizations": ["llm_stats"], "rank": 217, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-vl?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-browsecomp-zh", "caveat": "A high-difficulty benchmark purpose-built to comprehensively evaluate LLM agents on the Chinese web, consisting of 289 multi-hop questions spanning 11 diverse domains including Film & TV, Technology, Medicine, and History. Questions are reverse-engineered from short, objective, and easily verifiable answers, requiring sophisticated reasoning and information reconciliation beyond basic retrieval. The benchmark addresses linguistic, infrastructural, and censorship-related complexities in Chinese web environments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "BrowseComp-zh", "organization_count": 1, "organizations": ["llm_stats"], "rank": 218, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/browsecomp-zh?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1146-bust", "caveat": "BUST is a comprehensive benchmark for evaluating synthetic text detectors, focusing on their effectiveness against outputs from various Large Language Models (LLMs). BUST 是一个综合基准，旨在评估合成文本检测器，BUST 使用多种指标来评估检测器，包括语言特征、可读性和作者态度。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/BUST"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "BUST", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 219, "released": "2024-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/BUST"}, {"aliases": [], "benchmark_id": "opencompass-1983-bytemorph", "caveat": "ByteMorph is a benchmark for instruction-guided image editing, focusing on evaluating models’ capabilities in handling non-rigid motions such as camera viewpoint changes, object deformations, human articulations, and complex interactions. ByteMorph 是一个面向指令驱动图像编辑的基准，专注于评估模型在处理非刚性运动（如相机视角变化、物体变形、人类动作和复杂交互）方面的能力。 该基准包括超过 600 万对高分辨率图像编辑样本，涵盖多种动态编辑场景，支持对模型在多种非刚性运动类型下的表现进行细粒度评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ByteMorph"], "document_share": 0.0008271298593879239, "domain": "创作", "name": "ByteMorph", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 220, "released": "2025-06-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ByteMorph"}, {"aliases": [], "benchmark_id": "llm-stats-c-eval", "caveat": "C-Eval is a comprehensive Chinese evaluation suite designed to assess advanced knowledge and reasoning abilities of foundation models in a Chinese context. It comprises 13,948 multiple-choice questions across 52 diverse disciplines spanning humanities, science, and engineering, with four difficulty levels: middle school, high school, college, and professional. The benchmark includes C-Eval Hard, a subset of very challenging subjects requiring advanced reasoning abilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "C-Eval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 221, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/c-eval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-496-c-eval", "caveat": "C-Eval is a comprehensive Chinese evaluation suite for foundation models. It consists of 13948 multi-choice questions spanning 52 diverse disciplines and four difficulty levels. C-Eval 是一个全面的中文基础模型评估套件。它包含了13948个多项选择题，涵盖了52个不同的学科和四个难度级别。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/C-Eval"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "C-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 222, "released": "2023-05-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/C-Eval"}, {"aliases": [], "benchmark_id": "opencompass-1780-c-faith", "caveat": "C-FAITH, a Chinese QA hallucination benchmark created from 1,399 knowledge documents obtained from web scraping, totaling 60,702 entries. C-FAITH，这是一个中国 QA 幻觉基准，由从网络抓取中获得的 1,399 份知识文档创建，总共 60,702 个条目。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/C-FAITH"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "C-FAITH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 223, "released": "2025-04-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/C-FAITH"}, {"aliases": [], "benchmark_id": "opencompass-514-c3", "caveat": "A free-form multiple-Choice Chinese machine reading Comprehension dataset (C3), containing 13,369 documents (dialogues or more formally written mixed-genre texts) and their associated 19,577 multiple-choice free-form questions collected from Chinese-as-a-second-language examinations 一个自由形式的多项选择中文机器阅读理解数据集（C3），包含13369篇文献（对话或更正式的混合体裁文本）及其相关的19577道自由选择题，这些问题都是从汉语作为第二语言的考试中收集到的", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/C3"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "C3", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 224, "released": "2019-04-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/C3"}, {"aliases": [], "benchmark_id": "opencompass-1052-calm", "caveat": "CaLM is the first comprehensive benchmark for evaluating the causal reasoning capabilities of language models. The CaLM framework establishes a foundational taxonomy consisting of four modules: causal target, adaptation, metric, and error. CaLM是上海人工智能实验室联合同济大学、上海交通大学、北京大学及商汤科技发布首个大模型因果推理开放评测体系。首次从因果推理角度提出评估框架，为AI研究者打造可靠评测工具，从而为推进大模型认知能力向人类水平看齐提供指标参考。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CaLM"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "CaLM", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 225, "released": "2024-05-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CaLM"}, {"aliases": [], "benchmark_id": "llm-stats-capture-the-flag-challenges", "caveat": "Capture-the-Flag Challenges is OpenAI's internal expansion of competitive, professional-level cybersecurity capture-the-flag tasks used to evaluate vulnerability identification and exploitation capability under the Preparedness Framework.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "Capture-the-Flag Challenges (Internal)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 226, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/capture-the-flag-challenges?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1975-causalvqa", "caveat": "CausalVQA tests causal reasoning in videos across five question types, and state-of-the-art multimodal models still trail human performance. CausalVQA 是面向视频问答的因果推理基准，涵盖反事实、假设、预判、规划、描述五类问题，强调真实物理场景。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CausalVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "CausalVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 227, "released": "2025-06-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CausalVQA"}, {"aliases": [], "benchmark_id": "llm-stats-cbnsl", "caveat": "Curriculum Learning of Bayesian Network Structures (CBNSL) benchmark for evaluating algorithms that learn Bayesian network structures from data using curriculum learning techniques. The benchmark uses networks from the bnlearn repository and evaluates structure learning performance using BDeu scoring metrics.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "CBNSL", "organization_count": 1, "organizations": ["llm_stats"], "rank": 228, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cbnsl?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cc-bench-v2-backend", "caveat": "CC-Bench-V2 Backend evaluates coding agents on backend development tasks, measuring practical engineering ability to implement server-side logic, APIs, and system components.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "CC-Bench-V2 Backend", "organization_count": 1, "organizations": ["llm_stats"], "rank": 229, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-backend?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cc-bench-v2-frontend", "caveat": "CC-Bench-V2 Frontend evaluates coding agents on frontend development tasks, measuring ability to build UI components, handle styling, and implement client-side logic.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "CC-Bench-V2 Frontend", "organization_count": 1, "organizations": ["llm_stats"], "rank": 230, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-frontend?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cc-bench-v2-repo", "caveat": "CC-Bench-V2 Repo Exploration evaluates coding agents on repository-level understanding and navigation, measuring ability to explore, comprehend, and work across entire codebases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "CC-Bench-V2 Repo Exploration", "organization_count": 1, "organizations": ["llm_stats"], "rank": 231, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-bench-v2-repo?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cc-ocr", "caveat": "A comprehensive OCR benchmark for evaluating Large Multimodal Models (LMMs) in literacy. Comprises four OCR-centric tracks: multi-scene text reading, multilingual text reading, document parsing, and key information extraction. Contains 39 subsets with 7,058 fully annotated images, 41% sourced from real applications. Tests capabilities including text grounding, multi-orientation text recognition, and detecting hallucination/repetition across diverse visual challenges.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "CC-OCR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 232, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cc-ocr?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1552-ceb", "caveat": "CEB evaluates LLM bias compositionally, featuring 11k samples characterized across bias types, social groups, and tasks. CEB是一个用于大型语言模型偏差的组成评估基准，引入了包含 11,004 个样本的组成评估基准，从偏差类型、社会群体和任务三个维度描述每个数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CEB"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "CEB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 233, "released": "2024-07-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CEB"}, {"aliases": [], "benchmark_id": "llm-stats-cfeval", "caveat": "CFEval benchmark for evaluating code generation and problem-solving capabilities", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "CFEval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 234, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cfeval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1907-cfinbench", "caveat": "We present CFinBench: a meticulously crafted, the most comprehensive evaluation benchmark to date, for assessing the financial knowledge of LLMs under Chinese context. 为了更加全面地探究大语言模型在中文财经领域的能力，本工作提出了目前为止量级最大的中文财经评测基准（CFinBench）。该数据集共包含99,100个评测样本，并包含单选题、多选题和判断题在内的三种题型。该工作对当前主流的大模型从四个维度进行了详细评测：财经学科基础、财经资格认证、财经从业实践、财经法律法规。数据集和测评代码均已开源。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CFinBench"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "CFinBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 235, "released": "2024-10-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CFinBench"}, {"aliases": [], "benchmark_id": "opencompass-1079-cflue", "caveat": "CFLUE is the Chinese Financial Language Understanding Evaluation benchmark, designed to assess the capability of LLMs across various dimensions. CFLUE 是中国金融语言理解评估基准，旨在评估大型语言模型（LLMs）在各个维度上的能力。具体而言，CFLUE 提供了针对知识评估和应用评估量身定制的数据集。在知识评估方面，它包含超过 38,000 道选择题及相关的解决方案解释。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CFLUE"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "CFLUE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 236, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CFLUE"}, {"aliases": [], "benchmark_id": "opencompass-1513-cg-bench", "caveat": "CG-Bench is meant for evaluating MLLMs' long video understanding, including 12,129 QA pairs from 1219 videos in 3 major question types: perception, reasoning, and hallucination. CG-Bench用于评估多模态大模型的长视频理解能力，基于1219个视频设计了12129个涵盖感知、推理和幻觉三种问题类型的QA对。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CG-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "CG-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 237, "released": "2024-12-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CG-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-charadessta", "caveat": "Charades-STA is a benchmark dataset for temporal activity localization via language queries, extending the Charades dataset with sentence temporal annotations. It contains 12,408 training and 3,720 testing segment-sentence pairs from videos with natural language descriptions and precise temporal boundaries for localizing activities based on language queries.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "CharadesSTA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 238, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/charadessta?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1141-charm", "caveat": "CHARM is the first benchmark for comprehensively and in-depth evaluating the commonsense reasoning ability of large language models (LLMs) in Chinese, which covers both globally known and Chinese-specific commonsense. CHARM 是首个全面深入评估大语言模型（LLMs）在中文中的常识推理能力的基准，涵盖了全球通用的常识和特定于中国的常识。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CHARM"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "CHARM", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 239, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CHARM"}, {"aliases": [], "benchmark_id": "llm-stats-chartmuseum", "caveat": "ChartMuseum is a chart question-answering benchmark of 1,162 expert-annotated questions over real-world chart images drawn from 184 sources, including academic figures, infographics, and unconventional chart designs. It specifically targets questions that require visual reasoning, such as comparing unlabeled visual elements, tracking trajectories, and judging spatial relationships.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ChartMuseum", "organization_count": 1, "organizations": ["llm_stats"], "rank": 240, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/chartmuseum?top_n=500"}, {"aliases": ["Chartography"], "benchmark_id": "chartography", "caveat": "Chart comprehension benchmark with tool access. Scores depend on context length and tool configuration.", "document_count": 1, "document_ids": ["model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0008271298593879239, "domain": "vision", "name": "Chartography", "organization_count": 1, "organizations": ["Z.ai"], "rank": 241, "released": "2025-06-01", "source": "model_reports", "url": "https://github.com/Chartography/Chartography"}, {"aliases": [], "benchmark_id": "llm-stats-chartqa", "caveat": "ChartQA is a large-scale benchmark comprising 9.6K human-written questions and 23.1K questions generated from human-written chart summaries, designed to evaluate models' abilities in visual and logical reasoning over charts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ChartQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 242, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-chartqapro", "caveat": "ChartQAPro is a challenging benchmark for question answering over diverse, real-world charts and infographics.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ChartQAPro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 243, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/chartqapro?top_n=500"}, {"aliases": ["CharXiv Reasoning", "CharXiv"], "benchmark_id": "charxiv_reasoning", "caveat": "Chart reasoning benchmark with tool access. Scores depend on context length and tool configuration.", "document_count": 1, "document_ids": ["model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0008271298593879239, "domain": "vision", "name": "CharXiv Reasoning", "organization_count": 1, "organizations": ["Z.ai"], "rank": 244, "released": "2025-06-01", "source": "model_reports", "url": "https://github.com/CharXiv/CharXiv"}, {"aliases": [], "benchmark_id": "llm-stats-charxiv-d", "caveat": "CharXiv-D is the descriptive questions subset of the CharXiv benchmark, designed to assess multimodal large language models' ability to extract basic information from scientific charts. It contains descriptive questions covering information extraction, enumeration, pattern recognition, and counting across 2,323 diverse charts from arXiv papers, all curated and verified by human experts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "CharXiv-D", "organization_count": 1, "organizations": ["llm_stats"], "rank": 245, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-d?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-charxiv-r", "caveat": "CharXiv-R is the reasoning component of the CharXiv benchmark, focusing on complex reasoning questions that require synthesizing information across visual chart elements. It evaluates multimodal large language models on their ability to understand and reason about scientific charts from arXiv papers through various reasoning tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "CharXiv-R", "organization_count": 1, "organizations": ["llm_stats"], "rank": 246, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/charxiv-r?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1534-chase-code", "caveat": "CHASE is a unified framework to synthetically generate challenging problems using LLMs without human involvement CHASE是一个无需人工参与的统一框架，用于合成生成具有挑战性的问题", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CHASE-Code"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "CHASE-Code", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 247, "released": "2025-02-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CHASE-Code"}, {"aliases": ["Chatbot Arena", "LMArena", "LMSYS Arena"], "benchmark_id": "chatbot_arena", "caveat": "Elo from real user traffic; sampling and prompt distribution are outside any vendor's control.", "document_count": 1, "document_ids": ["model_reports:meta_llama_4"], "document_share": 0.0008271298593879239, "domain": "human_preference", "name": "Chatbot Arena", "organization_count": 1, "organizations": ["Meta"], "rank": 248, "released": "2023-05-03", "source": "model_reports", "url": "https://lmarena.ai/"}, {"aliases": [], "benchmark_id": "opencompass-692-chembench", "caveat": "ChemBench is a large-scale chemistry competency evaluation benchmark for language models, which includes nine chemistry core tasks and 4100 high-quality single-choice questions and answers. ChemBench是一个包含了九项化学核心任务，4100个高质量单选问答的大语言模型化学能力评测基准.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ChemBench"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "ChemBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 249, "released": "2024-02-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ChemBench"}, {"aliases": [], "benchmark_id": "llm-stats-chexpert-cxr", "caveat": "CheXpert is a large dataset of 224,316 chest radiographs from 65,240 patients for automated chest X-ray interpretation. The dataset includes uncertainty labels for 14 medical observations extracted from radiology reports. It serves as a benchmark for developing and evaluating automated chest radiograph interpretation models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "healthcare", "name": "CheXpert CXR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 250, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/chexpert-cxr?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-505-chid", "caveat": "CHID is a chinese idiom reading comprehension task, which requires to select the correct idiom to fill in the blank according to the context, with 10 candidate idioms. CHID是一个中文成语阅读理解任务，要求根据上下文选择正确的成语填空，共有10个候选成语。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CHID"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "CHID", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 251, "released": "2019-06-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CHID"}, {"aliases": [], "benchmark_id": "opencompass-1278-chronomagic-bench", "caveat": "ChronoMagic-Bench can evaluate the temporal and metamorphic capabilities of the T2V (text-to-video) models in time-lapse video generation, introducing 1,649 prompts and real-world videos as references. ChronoMagic-Bench用来评估 T2V （文本到视频 ）模型在延时视频生成中的时间和变形能力，引入了1649个提示和真实世界的视频作为参考。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ChronoMagic-Bench"], "document_share": 0.0008271298593879239, "domain": "创作", "name": "ChronoMagic-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 252, "released": "2024-06-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ChronoMagic-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-ci-memories-coverage", "caveat": "CI Memories measures privacy behavior in memory-enabled agents using contextual-integrity scenarios. This metric reports evaluation coverage.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500"], "document_share": 0.0008271298593879239, "domain": "memory", "name": "CI Memories Coverage", "organization_count": 1, "organizations": ["llm_stats"], "rank": 253, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-coverage?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ci-memories-violation", "caveat": "CI Memories measures privacy failures in memory-enabled agents using contextual-integrity scenarios. This metric is the violation rate; lower is better.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500"], "document_share": 0.0008271298593879239, "domain": "memory", "name": "CI Memories Violation Rate", "organization_count": 1, "organizations": ["llm_stats"], "rank": 254, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ci-memories-violation?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cl-bench", "caveat": "CL-bench is an open-source benchmark with its own data and rubrics for evaluating models on coding and agentic tasks, scored using a setup fully aligned with the official procedure.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "CL-bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 255, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cl-bench-life", "caveat": "CL-bench Life variant.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "CL-bench (Life)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 256, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cl-bench-%28life%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-claw-eval", "caveat": "Claw-Eval tests real-world agentic task completion across complex multi-step scenarios, evaluating a model's ability to use tools, navigate environments, and complete end-to-end tasks autonomously.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Claw-Eval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 257, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-claw-eval-mm", "caveat": "ClawEval-MM is the multimodal variant of ClawEval, evaluating agentic problem solving with visual inputs.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ClawEval-MM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 258, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/claw-eval-mm?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1849-cleanpatrick", "caveat": "CleanPatrick is a large-scale, real-world image data-cleaning benchmark with 496,377 binary annotations from 933 medical crowd workers for ranking off-topic, near-duplicate, and label-error issues. CleanPatrick是首个大规模图像数据清洗基准，基于公开的Fitzpatrick17k皮肤科数据集构建。该基准包含超过50万条来自933名医学众包工人的二元注释，涵盖三种数据质量问题：离题样本、近似重复样本和标签错误。通过医学专家验证，CleanPatrick提供了高质量的基准数据，用于评估图像数据清洗策略。基准测试结果表明，现有的数据清洗方法在近似重复检测中表现出色，但在标签错误检测方面仍面临挑战。CleanPatrick为数据清洗方法提供了标准化的评估框架，推动了更可靠的数据中心人工智能的发展。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CleanPatrick"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "CleanPatrick", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 259, "released": "2025-05-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CleanPatrick"}, {"aliases": [], "benchmark_id": "opencompass-2127-clembench", "caveat": "clembench is a benchmark framework for evaluating LLMs through dialogue game-based interactions, assessing chat-optimized models as conversational agents via multi-turn interactive game scenario. clembench是一个基于对话游戏的大语言模型评测框架，通过多回合交互式游戏场景评测聊天优化模型的会话智能体能力，包含Wordle、Taboo等多种游戏任务，评测指令遵循、目标导向行为和精细化交互理解等核心维度，提供可重复、可控制的自玩评测环境和综合得分机制，支持英语及多语言扩展的开源评测平台。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/clembench"], "document_share": 0.0008271298593879239, "domain": "创作", "name": "clembench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 260, "released": "2025-07-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/clembench"}, {"aliases": [], "benchmark_id": "opencompass-1834-clever", "caveat": "CLEVER is a benchmark suite for end-to-end code generation and formal verification in Lean 4, adapted from the HumanEval dataset. It requires models to generate implementations, formal specifications, and proofs—all verifiable by Lean's type checker, moving beyond test-case-driven evaluation. CLEVER is a benchmark suite for end-to-end code generation and formal verification in Lean 4, adapted from the HumanEval dataset. It requires models to generate implementations, formal specifications, and proofs—all verifiable by Lean's type checker, moving beyond test-case-driven evaluation.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CLEVER"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "CLEVER", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 261, "released": "2025-05-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CLEVER"}, {"aliases": [], "benchmark_id": "opencompass-1954-climateviz", "caveat": "ClimateViz is a large-scale multimodal benchmark designed to evaluate the scientific fact-checking and statistical reasoning capabilities of large language and vision-language models. It focuses on real-world climate science data, with over 49,000 high quality natural language claims. ClimateViz 是一个多模态基准数据集，用于评估大模型在气候科学图表上的事实核查与统计推理能力。数据来源于 NOAA、英国气象局等权威机构，共包含约 2,800 张科学图表与近 5 万条主张，标注为支持（support）、反驳（refute）或信息不足（NEI）。\n\n该数据集支持三种输入格式：图表+主张、表格+主张、图表标题+表格+主张，涵盖趋势识别、时空推理与科学对比等核心任务。ClimateViz 适用于多模态大模型和语言模型的系统性评估。\n图表转表格 + 图表标题 + 主张（Caption + Table", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ClimateViz"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ClimateViz", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 262, "released": "2025-06-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ClimateViz"}, {"aliases": [], "benchmark_id": "llm-stats-cloningscenarios", "caveat": "CloningScenarios is an expert-level multi-step reasoning benchmark about difficult genetic cloning scenarios in multiple-choice format. It evaluates dual-use biological knowledge relevant to bioweapons development.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CloningScenarios", "organization_count": 1, "organizations": ["llm_stats"], "rank": 263, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cloningscenarios?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cluewsc", "caveat": "CLUEWSC2020 is the Chinese version of the Winograd Schema Challenge, part of the CLUE benchmark. It focuses on pronoun disambiguation and coreference resolution, requiring models to determine which noun a pronoun refers to in a sentence. The dataset contains 1,244 training samples and 304 development samples extracted from contemporary Chinese literature.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CLUEWSC", "organization_count": 1, "organizations": ["llm_stats"], "rank": 264, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cluewsc?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1137-cmb", "caveat": "CMB is Comprehensive Medical Benchmark in Chinese, designed and rooted entirely within the native Chinese linguistic and cultural framework. CMB 是一个综合医学基准，专为中文而设计，并完全依赖于本土的中文语言和文化框架中。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CMB"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "CMB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 265, "released": "2024-04-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CMB"}, {"aliases": [], "benchmark_id": "llm-stats-cmmlu", "caveat": "CMMLU (Chinese Massive Multitask Language Understanding) is a comprehensive Chinese benchmark that evaluates the knowledge and reasoning capabilities of large language models across 67 different subject topics. The benchmark covers natural sciences, social sciences, engineering, and humanities with multiple-choice questions ranging from basic to advanced professional levels.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CMMLU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 266, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cmmlu?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-499-cmmlu", "caveat": "CMMLU is a comprehensive Chinese evaluation benchmark specifically designed to assess the knowledge and reasoning abilities of language models in the context of the Chinese language. CMMLU covers 67 topics ranging from basic subjects to advanced professional levels. It includes tasks that require calculations and reasoning in natural sciences, as well as tasks involving knowledge from humanities, social sciences, and practical aspects like Chinese driving rules. Moreover, many tasks within CMMLU have answers specific to China, which might not be universally applicable in other regions or languages. As a result, CMMLU serves as a fully localized Chinese evaluation benchmark. CMMLU是一个综合性的中文评估基准，专门用于评估语言模型在中文语境下的知识和推理能力。CMMLU涵盖了从基础学科到高级专业水平的67个主题。它包括：需要计算和推理的自然科学，需要知识的人文科学和社会科学,以及需要生活常识的中国驾驶规则等。此外，CMMLU中的许多任务具有中国特定的答案，可能在其他地区或语言中并不普遍适用。因此是一个完全中国化的中文测试基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CMMLU"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "CMMLU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 267, "released": "2023-06-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CMMLU"}, {"aliases": [], "benchmark_id": "opencompass-524-cmnli", "caveat": "CMNLI  is a Chinese natural language inference task, which requires to determine the logical relation between two sentences, with three relations: entailment, contradiction and neutral. CMNLI是一个中文自然语言推理任务，要求根据两个句子判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CMNLI"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "CMNLI", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 268, "released": null, "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CMNLI"}, {"aliases": [], "benchmark_id": "llm-stats-cmt-benchmark", "caveat": "CMT-Benchmark evaluates models on condensed matter theory problems, testing advanced physics reasoning across areas such as many-body systems, quantum field theory, and statistical mechanics.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500"], "document_share": 0.0008271298593879239, "domain": "physics", "name": "CMT-Benchmark", "organization_count": 1, "organizations": ["llm_stats"], "rank": 269, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cmt-benchmark?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cnmo-2024", "caveat": "China Mathematical Olympiad 2024 - A challenging mathematics competition.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "CNMO 2024", "organization_count": 1, "organizations": ["llm_stats"], "rank": 270, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cnmo-2024?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1546-codecriticbench", "caveat": "CodeCriticBench assesses LLMs' critiquing ability in code generation and QA tasks. Covering 10 criteria, it features a 4.3k-samples dataset with three difficulty levels and balanced distribution. CodeCriticBench ，旨在系统地评估LLMs在代码生成和代码问答任务中的批评能力。其涵盖 10 个不同的标准，数据集根据难度分为三个等级，共包含4.3k个样本，确保了难度级别的平衡分布。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CodeCriticBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "CodeCriticBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 271, "released": "2025-02-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CodeCriticBench"}, {"aliases": [], "benchmark_id": "opencompass-1624-codeelo", "caveat": "Researchers introduce CodeElo, a standardized competition-level code generation benchmark that effectively addresses all these challenges for the first time. CodeElo benchmark is mainly based on the official CodeForces platform and tries to align with the platform as much as possible. CodeElo，这是一个标准化的竞赛级代码生成基准测试，有效解决了所有这些挑战。CodeElo 基准测试主要基于官方 CodeForces 平台，并尽可能与该平台保持一致。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CodeElo"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "CodeElo", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 272, "released": "2025-01-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CodeElo"}, {"aliases": [], "benchmark_id": "llm-stats-codeforces", "caveat": "A competitive programming benchmark using problems from the CodeForces platform. The benchmark evaluates code generation capabilities of LLMs on algorithmic problems with difficulty ratings ranging from 800 to 2400. Problems cover diverse algorithmic categories including dynamic programming, graph algorithms, data structures, and mathematical problems with standardized evaluation through direct platform submission.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "CodeForces", "organization_count": 1, "organizations": ["llm_stats"], "rank": 273, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/codeforces?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-codegolf-v2-2", "caveat": "Codegolf v2.2 benchmark", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "Codegolf v2.2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 274, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/codegolf-v2.2?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1582-codemmlu", "caveat": "CodeMMLU is a comprehensive benchmark designed to evaluate the capabilities of large language models (LLMs) in coding and software knowledge. It builds upon the structure of multiple-choice question answering (MCQA) to cover a wide range of programming tasks and domains. CodeMMLU 是一个旨在评估大型语言模型（LLMs）在编码和软件知识方面能力的全面基准。它基于多项选择题回答（MCQA）的结构，涵盖了广泛的编程任务和领域，包括代码生成、缺陷检测、软件工程原则等。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CodeMMLU"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "CodeMMLU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 275, "released": "2024-06-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CodeMMLU"}, {"aliases": [], "benchmark_id": "llm-stats-cohere-agentic-question-answering", "caveat": "Cohere's internal North evaluation for measuring how well a model answers enterprise questions using MCP-connected cloud file systems. Scores are reported with LLM-as-a-judge techniques.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500"], "document_share": 0.0008271298593879239, "domain": "question_answering", "name": "Cohere Agentic Question Answering", "organization_count": 1, "organizations": ["llm_stats"], "rank": 276, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-agentic-question-answering?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cohere-data-analysis", "caveat": "Cohere's internal North evaluation for measuring a model's ability to perform data science tasks over uploaded spreadsheets. Scores are reported with LLM-as-a-judge techniques.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Cohere Data Analysis", "organization_count": 1, "organizations": ["llm_stats"], "rank": 277, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-data-analysis?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cohere-memory-usage-quality", "caveat": "Cohere's internal North evaluation for measuring how well an agent uses information from North's memory system across sessions. Scores are reported with LLM-as-a-judge techniques.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500"], "document_share": 0.0008271298593879239, "domain": "memory", "name": "Cohere Memory Usage Quality", "organization_count": 1, "organizations": ["llm_stats"], "rank": 278, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cohere-memory-usage-quality?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-collie", "caveat": "COLLIE is a grammar-based framework for systematic construction of constrained text generation tasks. It allows specification of rich, compositional constraints across diverse generation levels and modeling challenges including language understanding, logical reasoning, and semantic planning. The COLLIE-v1 dataset contains 2,080 instances across 13 constraint structures.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "COLLIE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 279, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/collie?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1250-collie", "caveat": "COLLIE is a grammar-based framework that allows the specification of rich, compositional constraints with diverse generation levels (word, sentence, paragraph, passage) and modeling challenges (e.g.,language understanding, logical reasoning, counting, semantic planning). COLLIE用于评估大模型在约束性文本生成任务中的表现，可指定具有不同生成级别（单词、句子、段落、段落）和建模挑战（例如，语言理解、逻辑推理、计数、语义规划）的丰富组合。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Collie"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "Collie", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 280, "released": "2023-07-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Collie"}, {"aliases": [], "benchmark_id": "opencompass-1777-colorbench", "caveat": "ColorBench, an innovative benchmark meticulously crafted to assess the capabilities of VLMs in color understanding, including color perception, reasoning, and robustness. ColorBench，这是一个创新且精心设计的基准测试，旨在评估VLMs在颜色理解方面的能力，包括颜色感知、推理和鲁棒性。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ColorBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ColorBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 281, "released": "2025-04-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ColorBench"}, {"aliases": [], "benchmark_id": "opencompass-1943-combibench", "caveat": "We introduce CombiBench, a comprehensive benchmark comprising 100 combinatorial problems, each formalized in Lean4 and paired with its corresponding informal statement. The problem set covers a wide spectrum of difficulty levels, ranging from middle school to IMO and university level. CombiBench是一个包含100个组合问题的综合基准测试，每个问题都用Lean4进行了形式化，并附有其对应的非形式化表述。这些问题涵盖了从中学生到国际数学奥林匹克竞赛（IMO）以及大学水平的广泛难度范围。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CombiBench"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "CombiBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 282, "released": "2025-04-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CombiBench"}, {"aliases": [], "benchmark_id": "llm-stats-common-voice-15", "caveat": "Common Voice is a massively-multilingual collection of transcribed speech intended for speech technology research and development. Version 15.0 contains 28,750 recorded hours across 114 languages, consisting of crowdsourced voice recordings with corresponding transcriptions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500"], "document_share": 0.0008271298593879239, "domain": "speech_to_text", "name": "Common Voice 15", "organization_count": 1, "organizations": ["llm_stats"], "rank": 283, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/common-voice-15?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-commonsenseqa", "caveat": "CommonSenseQA is a multiple-choice question answering dataset that requires different types of commonsense knowledge to predict correct answers. It contains 12,102 questions with one correct answer and four distractors, designed to test semantic reasoning and conceptual relationships. Questions are created based on ConceptNet concepts and require prior world knowledge for accurate reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CommonSenseQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 284, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/commonsenseqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-511-commonsenseqa", "caveat": "CommonsenseQA is a multiple-choice question answering dataset that requires different types of commonsense knowledge to predict the correct answers . It contains 12,102 questions with one correct answer and four distractor answers. CommonsenseQA是一个选择题数据集，它需要不同类型的常识知识来预测正确答案。它包含12,102个问题，有一个正确答案和四个干扰答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CommonSenseQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "CommonSenseQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 285, "released": "2018-11-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CommonSenseQA"}, {"aliases": [], "benchmark_id": "opencompass-1336-compbench", "caveat": "CompBench is designed to evaluate the comparative reasoning capability of multimodal large language models, including a collection of around 40K image pairs and visually oriented questions covering 8 dimensions CompBench旨在评估多模态大模型的比较推理能力，包含约4万个图像对及8个维度的视觉配对问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CompBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "CompBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 286, "released": "2024-07-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CompBench"}, {"aliases": [], "benchmark_id": "llm-stats-complexfuncbench", "caveat": "ComplexFuncBench is a benchmark designed to evaluate large language models' capabilities in handling complex function calling scenarios. It encompasses multi-step and constrained function calling tasks that require long-parameter filling, parameter value reasoning, and managing contexts up to 128k tokens. The benchmark includes 1,000 samples across five real-world scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "ComplexFuncBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 287, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/complexfuncbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-openai-connectors", "caveat": "Connectors is an OpenAI internal production benchmark measuring reliable use of connector-based tools in agentic workflows, reported as a pass rate.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Connectors", "organization_count": 1, "organizations": ["llm_stats"], "rank": 288, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-connectors?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1679-contextualjudgebench", "caveat": "ContextualJudgeBench is a pairwise benchmark with 2,000 samples for evaluating LLM-as-judge models in two contextual settings: Contextual QA and summarization. We propose a pairwise evaluation hierarchy and generate splits for our proposed hierarchy. ContextualJudgeBench 是一个包含 2,000 个样本的成对基准，用于评估在两个上下文环境下的LLM-as-judge 模型：上下文问答和摘要。我们提出一个成对评估层次结构，并为我们的层次结构生成分割。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ContextualJudgeBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "ContextualJudgeBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 289, "released": "2025-03-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ContextualJudgeBench"}, {"aliases": [], "benchmark_id": "llm-stats-contphy", "caveat": "ContPhy is a continuum physical-reasoning benchmark evaluating understanding of physical dynamics in video.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ContPhy", "organization_count": 1, "organizations": ["llm_stats"], "rank": 290, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/contphy?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1269-convbench", "caveat": "ConvBench is a novel multi-turn conversation evaluation benchmark tailored for Large Vision-Language Models (LVLMs). It comprises 577 meticulously curated multi-turn conversations encompassing 215 tasks reflective of real-world demands. ConvBench是专用于大型视觉语言模型 （LVLM）的新型多轮对话评估基准，由基于215 个反映实际需求的任务的577个多轮次对话组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ConvBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "ConvBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 291, "released": "2024-03-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ConvBench"}, {"aliases": [], "benchmark_id": "opencompass-529-copa", "caveat": "COPA is a causal inference task, which requires to select the correct causal relation based on the given premise. COPA是一个因果推断任务，要求根据给定的前提，选择正确的因果关系。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/COPA"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "COPA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 292, "released": null, "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/COPA"}, {"aliases": [], "benchmark_id": "opencompass-1622-coral", "caveat": "Researchers present a large-scale conversational RAG benchmark named CORAL and propose a unified framework for standardizing and evaluating various conversational RAG baselines. CORAL 是一个大规模对话 RAG 基准，包含一个统一框架，用于标准化和评估各种对话 RAG 基线。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CORAL"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "CORAL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 293, "released": "2024-10-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CORAL"}, {"aliases": [], "benchmark_id": "llm-stats-corpusqa", "caveat": "CorpusQA is a multi-document, free-form long-context question answering benchmark in which a model must retrieve and reason over information distributed across a large corpus to produce open-ended answers that are scored by an LLM judge.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "CorpusQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 294, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-corpusqa-1m", "caveat": "CorpusQA 1M is a long-context question answering benchmark designed to evaluate models at approximately 1 million token contexts. Models are scored on accuracy when retrieving and reasoning over information distributed across an extremely long input corpus.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "CorpusQA 1M", "organization_count": 1, "organizations": ["llm_stats"], "rank": 295, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/corpusqa-1m?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-countbench", "caveat": "CountBench evaluates object counting capabilities in visual understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CountBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 296, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/countbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-countqa", "caveat": "CountQA is a benchmark for visual object counting and quantity reasoning over images.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "CountQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 297, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/countqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-covost2", "caveat": "CoVoST 2 is a large-scale multilingual speech translation corpus derived from Common Voice, covering translations from 21 languages into English and from English into 15 languages. The dataset contains 2,880 hours of speech with 78K speakers for speech translation research.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "speech_to_text", "name": "CoVoST2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 298, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-covost2-en-zh", "caveat": "CoVoST 2 English-to-Chinese subset is part of the large-scale multilingual speech translation corpus derived from Common Voice. This subset focuses specifically on English to Chinese speech translation tasks within the broader CoVoST 2 dataset.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500"], "document_share": 0.0008271298593879239, "domain": "speech_to_text", "name": "CoVoST2 en-zh", "organization_count": 1, "organizations": ["llm_stats"], "rank": 299, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/covost2-en-zh?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-coworkbench", "caveat": "CoWorkBench is Qwen's internal cowork benchmark for evaluating long-horizon office and productivity agent tasks across domains such as computer science, finance, law, and medicine.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "productivity", "name": "CoWorkBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 300, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/coworkbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1972-cpret", "caveat": "Programming contests are long used to evaluate algorithmic thinking and coding skills, and have recently become benchmarks for assessing large language models (LLMs). However, the rapid expansion of problem sets has led to a surge in duplicate or highly similar problems, compromising fairness ... 编程竞赛长期用于评估算法与编程能力，近年来也被用于大语言模型的评测。但随着题库扩展，重复或相似题激增，影响竞赛公平性与模型评测效果。为此，本文提出“相似题目检索”任务，并构建统一检索基准数据集 CPRet，涵盖题目与代码的四类检索任务，包含自动爬取和人工标注的数据样本。同时，设计并训练了两种检索模型 CPRetriever-Code 与 CPRetriever-Prob，显著提升检索效果。实验还发现相似题会提高模型得分、减小模型差异，强调了评测中引入“相似性感知”的必要性。我们还发布了开源检索平台，支持重复题检测与相似题推荐。项目地址：https://github.com/coldchair/", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CPRet"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "CPRet", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 301, "released": "2025-05-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CPRet"}, {"aliases": [], "benchmark_id": "llm-stats-crag", "caveat": "CRAG (Comprehensive RAG Benchmark) is a factual question answering benchmark consisting of 4,409 question-answer pairs across 5 domains (finance, sports, music, movie, open domain) and 8 question categories. The benchmark includes mock APIs to simulate web and Knowledge Graph search, designed to represent the diverse and dynamic nature of real-world QA tasks with temporal dynamism ranging from years to seconds. It evaluates retrieval-augmented generation systems for trustworthy question answering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CRAG", "organization_count": 1, "organizations": ["llm_stats"], "rank": 302, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/crag?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1626-crag", "caveat": "The Comprehensive RAG Benchmark (CRAG) is a rich and comprehensive factual question answering benchmark designed to advance research in RAG. Besides question-answer pairs, CRAG provides mock APIs to simulate web and knowledge graph search. CRAG是一个丰富且全面的基于事实的问题回答基准，旨在推进 RAG 研究。除了问答对之外，CRAG 还提供了模拟网页和知识图谱搜索的模拟 API。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CRAG"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "CRAG", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 303, "released": "2024-06-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CRAG"}, {"aliases": [], "benchmark_id": "opencompass-1643-creation-mmbench", "caveat": "A multimodal benchmark specifically designed to evaluate the creative capabilities of MLLMs. It features three main aspects: 1. Comprehensive Creation Benchmark for MLLM and LLM. 2. Robust Evaluation Methodology. 3. Attractive Experiment Insight. 专为评估 多模态大模型 的创作能力而设计的多模态基准。采用两个不同指标对模型的基础感知能力和深层次视觉创作能力进行评估，采用GPT-4o作为评判模型进行评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Creation-MMBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "Creation-MMBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 304, "released": "2025-03-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Creation-MMBench"}, {"aliases": [], "benchmark_id": "llm-stats-creative-writing-v3", "caveat": "EQ-Bench Creative Writing v3 is an LLM-judged creative writing benchmark that evaluates models across 32 writing prompts with 3 iterations per prompt. Uses a hybrid scoring system combining rubric assessment and Elo ratings through pairwise comparisons. Challenges models in areas like humor, romance, spatial awareness, and unique perspectives to assess emotional intelligence and creative writing capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500"], "document_share": 0.0008271298593879239, "domain": "creativity", "name": "Creative Writing v3", "organization_count": 1, "organizations": ["llm_stats"], "rank": 305, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/creative-writing-v3?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-creativework", "caveat": "CreativeWork evaluates agents on open-ended creative production tasks within realistic tool and application environments, measuring the quality and completeness of generated deliverables.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CreativeWork", "organization_count": 1, "organizations": ["llm_stats"], "rank": 306, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/creativework?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2046-crew-wildfire", "caveat": "CREW-WILDFIRE is a benchmark designed to evaluate large language model-based multi-agent systems’ collaboration capabilities in complex, dynamic tasks, targeting agentic frameworks with perception, planning, and execution abilities. CREW-WILDFIRE 是一个用于评估基于大语言模型的多智能体系统在复杂动态任务中协作能力的基准，面向具备感知、规划与执行能力的智能体框架。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CREW-WILDFIRE"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "CREW-WILDFIRE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 307, "released": "2025-07-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CREW-WILDFIRE"}, {"aliases": [], "benchmark_id": "opencompass-568-criticbench", "caveat": "CriticBench, a novel benchmark designed to comprehensively and reliably evaluate four key critique ability dimensions of LLMs: feedback, comparison, refinement and meta-feedback. CriticBench encompasses nine diverse tasks, each assessing the LLMs' ability to critique responses at varying levels of quality granularity. CriticBench是一个新颖的基准，旨在全面可靠地评估LLM的四个关键批判能力维度。CriticBench包括九项不同的任务，每项任务都评估LLM在不同质量粒度水平上对响应进行批评的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CriticBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "CriticBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 308, "released": "2024-02-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CriticBench"}, {"aliases": [], "benchmark_id": "artificial-analysis-critpt", "caveat": "Physics reasoning", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/critpt"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "CritPt", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 309, "released": "2025-11-21", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/critpt"}, {"aliases": [], "benchmark_id": "llm-stats-critpt", "caveat": "CritPT is a challenging reasoning benchmark reported by Qwen for evaluating frontier mathematical and critical problem-solving capability.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "CritPT", "organization_count": 1, "organizations": ["llm_stats"], "rank": 310, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/critpt?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-crossvid", "caveat": "CrossVid evaluates cross-video reasoning, requiring models to integrate information across multiple videos.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "CrossVid", "organization_count": 1, "organizations": ["llm_stats"], "rank": 311, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/crossvid?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1739-crosswordbench", "caveat": "CrossWordBench, a benchmark designed to evaluate the reasoning capabilities of both LLMs and LVLMs through the medium of crossword puzzles. CrossWordBench，这是一个基准测试，旨在通过填字游戏的方式来评估LLMs和LVLMs的推理能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CrossWordBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "CrossWordBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 312, "released": "2025-03-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CrossWordBench"}, {"aliases": [], "benchmark_id": "opencompass-1123-crows-pairs", "caveat": "CrowS-Pairs has 1508 examples that cover stereotypes dealing with nine types of bias, like race, religion, and age. In CrowS-Pairs a model is presented with two sentences: one that is more stereotyping and another that is less stereotyping. CrowS-Pairs 包含 1508 个示例，涵盖与九种偏见类型相关的刻板印象，例如种族、宗教和年龄。在 CrowS-Pairs 中，模型会接收到两句话：一句是更具刻板印象的，另一句则是较少刻板印象的。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Crows-Pairs"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "Crows-Pairs", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 313, "released": "2020-09-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Crows-Pairs"}, {"aliases": [], "benchmark_id": "opencompass-1500-crpe", "caveat": "CRPE is a benchmark designed to quantitatively evaluate the object recognition and relation comprehension ability of models. It consists of four splits, and the evaluation is formulated as single-choice questions. CRPE用于定量评估多模态大模型的对象识别和关系理解能力，分为四部分，以单选题形式呈现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CRPE"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "CRPE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 314, "released": "2024-02-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CRPE"}, {"aliases": [], "benchmark_id": "llm-stats-crperelation", "caveat": "Clinical reasoning problems evaluation benchmark for assessing diagnostic reasoning and medical knowledge application capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CRPErelation", "organization_count": 1, "organizations": ["llm_stats"], "rank": 315, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/crperelation?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-crux-o", "caveat": "CRUXEval-O (output prediction) is part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate AI models' capabilities in code reasoning, understanding, and execution. The benchmark tests models' ability to predict correct function outputs given function code and inputs, focusing on short problems that a good human programmer should be able to solve in a minute.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CRUX-O", "organization_count": 1, "organizations": ["llm_stats"], "rank": 316, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/crux-o?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cruxeval-input-cot", "caveat": "CRUXEval input prediction task with Chain of Thought (CoT) prompting. Part of the CRUXEval benchmark for code reasoning, understanding, and execution evaluation. Given a Python function and its expected output, the task is to predict the appropriate input using chain-of-thought reasoning. Consists of 800 Python functions (3-13 lines) designed to evaluate code comprehension and reasoning capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CRUXEval-Input-CoT", "organization_count": 1, "organizations": ["llm_stats"], "rank": 317, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-input-cot?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cruxeval-o", "caveat": "CruxEval-O is the output prediction task of the CRUXEval benchmark, designed to evaluate code reasoning, understanding, and execution capabilities. It consists of 800 Python functions (3-13 lines) where models must predict the output given a function and input. The benchmark tests fundamental code execution reasoning abilities and goes beyond simple code generation to assess deeper understanding of program behavior.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CruxEval-O", "organization_count": 1, "organizations": ["llm_stats"], "rank": 318, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-o?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cruxeval-output-cot", "caveat": "CRUXEval-O (output prediction) with Chain-of-Thought prompting. Part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate code reasoning, understanding, and execution capabilities. The output prediction task requires models to predict the output of a given Python function with specific inputs, evaluated using chain-of-thought reasoning methodology.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "CRUXEval-Output-CoT", "organization_count": 1, "organizations": ["llm_stats"], "rank": 319, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cruxeval-output-cot?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-924-cs-bench", "caveat": "CS-Bench, the first bilingual (Chinese-English) benchmark dedicated to evaluating the performance of LLMs in computer science. CS-Bench comprises approximately 5K meticulously curated test samples, covering 26 subfields across 4 key areas of computer science, encompassing various task forms and divisions of knowledge and reasoning. CS-Bench 第一个专门用于评估LLMs在计算机科学中表现的双语（中英文）基准。CS-Bench包括约5,000个精心策划的测试样本，涵盖了计算机科学4个关键领域中的26个子领域，并包括各种任务形式和知识推理的划分。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CS-Bench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "CS-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 320, "released": "2024-06-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CS-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1219-cs-eval", "caveat": "CS-Eval is a large language model cybersecurity capability evaluation suite jointly established by Alibaba Security, Fudan University, and the University of Chinese Academy of Sciences. The dataset encompasses 11 major cybersecurity categories and 42 subcategories, offering comprehensive assessment CS-Eval 是由阿里安全、复旦大学和中国科学院大学联合建立的大模型网络安全能力评测集。数据集覆盖11个网络安全大类领域、42个子类领域，提供知识型和实战型的综合评估任务，支持用户自主评测，同时为大模型落地网络安全提供参考和启发。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CS-Eval"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "CS-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 321, "released": "2024-05-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CS-Eval"}, {"aliases": [], "benchmark_id": "llm-stats-csimpleqa", "caveat": "Chinese SimpleQA is the first comprehensive Chinese benchmark to evaluate the factuality ability of language models to answer short questions. It contains 3,000 high-quality questions spanning 6 major topics with 99 diverse subtopics, designed to assess Chinese factual knowledge across humanities, science, engineering, culture, and society.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "CSimpleQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 322, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/csimpleqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-519-csl", "caveat": "CSL is a large-scale Chinese Scientific Literature dataset, which contains the titles, abstracts, keywords and academic fields of 396k papers. CSL是一个大规模的中文科技文献数据集，包含 39.6 万篇论文的标题、摘要、关键词和学术领域信息。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CSL"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "CSL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 323, "released": "2022-09-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CSL"}, {"aliases": [], "benchmark_id": "opencompass-1833-csts", "caveat": "CSTS is a synthetic benchmark for evaluating correlation structure discovery in multivariate time series. It features 23 distinct correlation structures with systematic data variations (distribution shifts, sparsification, downsampling) and provides ground truth labels for validation. CSTS is a synthetic benchmark for evaluating correlation structure discovery in multivariate time series. It features 23 distinct correlation structures with systematic data variations (distribution shifts, sparsification, downsampling) and provides ground truth labels for validation.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CSTS"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "CSTS", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 324, "released": "2025-05-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CSTS"}, {"aliases": [], "benchmark_id": "opencompass-1268-ctibench", "caveat": "CTIBench is a benchmark designed to assess LLMs' performance in CTI (Cyber threat intelligence) applications. It includes multiple datasets focused on evaluating knowledge acquired by LLMs in the cyber-threat landscape. CTIBench旨在评估LLM在CTI（网络安全情报）场景下的能力，包含多个数据集，专注于评测LLM在网络威胁环境中获得的知识。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CTIBench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "CTIBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 325, "released": "2024-06-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CTIBench"}, {"aliases": [], "benchmark_id": "llm-stats-cursorbench-3-2", "caveat": "CursorBench v3.2 evaluates coding agents on interactive software engineering tasks in the Cursor environment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "CursorBench v3.2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 326, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cursorbench-3.2?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1985-cvdp", "caveat": "CVDP is a next-generation benchmark for evaluating large language models (LLMs) and agents in hardware design and verification, comprising 783 problems across 13 task categories, including RTL generation, verification, debugging, specification alignment, and technical Q&A. CVDP 是一个面向大型语言模型（LLM）和智能体的下一代硬件设计与验证评测基准，涵盖 13 类任务共 783 个问题，涉及 RTL 生成、验证、调试、规范对齐和技术问答等。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CVDP"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "CVDP", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 327, "released": "2025-06-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CVDP"}, {"aliases": ["CVE-Bench", "CVEBench"], "benchmark_id": "cvebench", "caveat": "Real CVE exploitation in sandboxes. Vendors report it as a safety ceiling rather than a capability to maximize.", "document_count": 1, "document_ids": ["model_reports:openai_gpt_5_6_system_card"], "document_share": 0.0008271298593879239, "domain": "security", "name": "CVE-Bench", "organization_count": 1, "organizations": ["OpenAI"], "rank": 328, "released": "2025-03-21", "source": "model_reports", "url": "https://github.com/uiuc-kang-lab/cve-bench"}, {"aliases": [], "benchmark_id": "opencompass-1239-cvqa", "caveat": "CVQA is a new culturally-diverse multilingual Visual Question Answering benchmark, designed to cover a rich set of languages and cultures. It includes culturally-driven images and 10k questions from across 30 countries on 4 continents. CVQA是一种新的文化多元化多语言视觉问答基准，旨在涵盖丰富的语言和文化，包括来自四大洲30个国家/地区的文化向图像和10000个问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "CVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 329, "released": "2024-06-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CVQA"}, {"aliases": [], "benchmark_id": "llm-stats-cvtg-2k", "caveat": "CVTG-2K (Chinese Visual Text Generation 2K) is a benchmark for evaluating text-to-image models on their ability to accurately render text within generated images. It measures Word Accuracy, Normalized Edit Distance (NED), and CLIPScore across 2,000 prompts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cvtg-2k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "image-generation", "name": "CVTG-2K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 330, "released": "2025-03-30", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cvtg-2k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cybench", "caveat": "CyBench is a suite of Capture-the-Flag (CTF) challenges measuring agentic cyber attack capabilities. It evaluates dual-use cybersecurity knowledge and measures the 'unguided success rate', where agents complete tasks end-to-end without guidance on appropriate subtasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "CyBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 331, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cybench?top_n=500"}, {"aliases": [], "benchmark_id": "cybergym", "caveat": "Real-world cybersecurity tasks; tool permissions and time limits are part of the result, not incidental to it.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "security", "name": "CyberGym", "organization_count": 1, "organizations": ["Tencent"], "rank": 332, "released": null, "source": "model_reports", "url": "https://openreview.net/forum?id=2YvbLQEdYt"}, {"aliases": [], "benchmark_id": "llm-stats-cybergym", "caveat": "CyberGym is a benchmark for evaluating AI agents on cybersecurity tasks, testing their ability to identify vulnerabilities, perform security analysis, and complete security-related challenges in a controlled environment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "CyberGym", "organization_count": 1, "organizations": ["llm_stats"], "rank": 333, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cybergym?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1349-cyberseceval", "caveat": "CyberSecEval is a comprehensive benchmark developed to help bolster the cybersecurity of LLMs. It provides a thorough evaluation in two crucial security domains: the propensity to generate insecure code and the level of compliance when asked to assist in cyberattacks. CyberSecEval旨在评估LLM的安全性，聚焦于大模型生成不安全代码的倾向以及当被要求协助网络攻击时的合规性水平。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/CyberSecEval"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "CyberSecEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 334, "released": "2023-12-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/CyberSecEval"}, {"aliases": [], "benchmark_id": "llm-stats-cyberseceval-4", "caveat": "CyberSecEval 4 is an evaluation suite covering cybersecurity-related capabilities and risks of large language models. The insecure-code-generation tracks measure whether a model produces vulnerable code: the Instruct track presents coding requests designed to elicit known insecure patterns, while the Autocomplete track prompts the model with code context leading up to a known insecure pattern, with vulnerabilities detected via static analysis.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "CyberSecEval 4", "organization_count": 1, "organizations": ["llm_stats"], "rank": 335, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cyberseceval-4?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-cybersecurity-ctfs", "caveat": "Cybersecurity Capture the Flag (CTF) benchmark for evaluating LLMs in offensive security challenges. Contains diverse cybersecurity tasks including cryptography, web exploitation, binary analysis, and forensics to assess AI capabilities in cybersecurity problem-solving.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "Cybersecurity CTFs", "organization_count": 1, "organizations": ["llm_stats"], "rank": 336, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/cybersecurity-ctfs?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2030-dabstep", "caveat": "DABstep is a benchmark for evaluating AI agents’ multi-step reasoning and planning abilities in realistic data analysis tasks, targeting code-executing language model agents. DABstep 是一个用于评估 AI 智能体在现实多步骤数据分析任务中推理与规划能力的基准，面向具备代码执行能力的语言模型代理。该基准包含450多个任务，源自金融分析平台，涵盖结构化数据处理、非结构化文档理解和跨源信息整合。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DABstep"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "DABstep", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 337, "released": "2025-06-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DABstep"}, {"aliases": [], "benchmark_id": "llm-stats-dailyomni", "caveat": "DailyOmni evaluates multimodal models on daily-life video understanding tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "DailyOmni", "organization_count": 1, "organizations": ["llm_stats"], "rank": 338, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/dailyomni?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1078-debugbench", "caveat": "DebugBench is an LLM debugging benchmark consisting of 4,253 instances. It covers four major bug categories and 18 minor types in C++, Java, and Python. DebugBench 是一个包含 4,253 个实例的 LLM 调试基准。它涵盖了 C++、Java 和 Python 中的四个主要错误类别和 18 种次要类型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DebugBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "DebugBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 339, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DebugBench"}, {"aliases": [], "benchmark_id": "llm-stats-deck-bench", "caveat": "DECK-Bench is Moonshot AI's internal evaluation of agents on creating and reasoning over presentation-style knowledge-work artifacts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "productivity", "name": "DECK-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 340, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/deck-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1605-deepfake-eval-2024", "caveat": "Deepfake-Eval-2024 is an in-the-wild deepfake dataset. Deepfake-Eval-2024 contains 44 hours of videos, 56.5 hours of audio, and 1,975 images, encompassing contemporary manipulation technologies, diverse media content, 88 different website sources, and 52 different languages. Deepfake-Eval-2024 是一个真实场景的深度伪造数据集。该数据集包含44小时的视频、56.5小时的音频和1,975张图像，涵盖当代篡改技术、多样化的媒体内容、来自88个不同网站来源的素材以及52种不同语言。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Deepfake-Eval-2024"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "Deepfake-Eval-2024", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 341, "released": "2025-03-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Deepfake-Eval-2024"}, {"aliases": [], "benchmark_id": "llm-stats-deep-planning", "caveat": "DeepPlanning evaluates LLMs on complex multi-step planning tasks requiring long-horizon reasoning, goal decomposition, and strategic decision-making.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "DeepPlanning", "organization_count": 1, "organizations": ["llm_stats"], "rank": 342, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/deep-planning?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1987-deepresearchbench", "caveat": "DeepResearch Bench is a comprehensive benchmark designed to evaluate large language model (LLM) agents on complex research tasks. DeepResearch Bench 是一个面向大型语言模型（LLM）智能体的综合性评测基准，专为评估其在复杂研究任务中的表现而设计。 该基准包含 100 个由 22 个领域的专家精心设计的博士级研究任务，涵盖多领域的深度研究需求。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DeepResearchBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "DeepResearchBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 343, "released": "2025-06-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DeepResearchBench"}, {"aliases": [], "benchmark_id": "llm-stats-deepsearchqa", "caveat": "DeepSearchQA is a benchmark for evaluating deep search and question-answering capabilities, testing models' ability to perform multi-hop reasoning and information retrieval across complex knowledge domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "DeepSearchQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 344, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepsearchqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-deepswe", "caveat": "DeepSWE is a software engineering agent benchmark evaluated with the mini-swe-agent harness, where each task is solved in an isolated container with no internet access. It measures an agent's ability to autonomously resolve real-world coding issues end to end.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "DeepSWE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 345, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-deepswe-1-0", "caveat": "DeepSWE 1.0 pass@1 benchmark as reported by Artificial Analysis using provider-specific harness runs.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "DeepSWE 1.0", "organization_count": 1, "organizations": ["llm_stats"], "rank": 346, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.0?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-deepswe-1-1", "caveat": "DeepSWE 1.1 evaluates software engineering agents using the mini-swe-agent harness.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "DeepSWE 1.1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 347, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/deepswe-1.1?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-dermmcqa", "caveat": "Dermatology multiple choice question assessment benchmark for evaluating medical knowledge and diagnostic reasoning in dermatological conditions and treatments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "healthcare", "name": "DermMCQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 348, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/dermmcqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-design2code", "caveat": "Design2Code evaluates the ability to generate code (HTML/CSS/JS) from visual designs.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Design2Code", "organization_count": 1, "organizations": ["llm_stats"], "rank": 349, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/design2code?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2032-dice-bench", "caveat": "DICE-BENCH is a benchmark for evaluating large language models’ tool-use capabilities in multi-round, multi-party dialogues, focusing on function selection, parameter filling, and dialogue context comprehension. DICE-BENCH 是一个评估大型语言模型在多轮、多方对话中工具调用能力的基准，涵盖函数选择、参数填充和对话上下文理解等维度。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DICE-BENCH"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "DICE-BENCH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 350, "released": "2025-06-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DICE-BENCH"}, {"aliases": [], "benchmark_id": "opencompass-1648-dme", "caveat": "VLB provides a robust and comprehensive assessment for LVLMs with reduced data contamination and flexible complexity. Based on LlavaBench and MMvet, we have curated two more challenging versions of the datasets: LlavaBench_hard and MMvet_hard. These are the hardest multimodal combinations. VLB 为 LVLMs 提供了一种稳健且全面的评估，降低了数据污染并具有灵活的复杂性。基于 LlavaBench 和 MMvet，我们精心制作了两个更具挑战性的数据集版本：LlavaBench_hard 和 MMvet_hard。这些是我们动态策略中最具挑战性的多模态组合（V1+L4）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DME"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "DME", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 351, "released": "2024-10-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DME"}, {"aliases": [], "benchmark_id": "llm-stats-docvqa", "caveat": "A dataset for Visual Question Answering on document images containing 50,000 questions defined on 12,000+ document images. The benchmark tests AI's ability to understand document structure and content, requiring models to comprehend document layout and perform information retrieval to answer questions about document images.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "DocVQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 352, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-docvqatest", "caveat": "DocVQA is a Visual Question Answering benchmark on document images containing 50,000 questions defined on 12,000+ document images. The benchmark focuses on understanding document structure and content to answer questions about various document types including letters, memos, notes, and reports from the UCSF Industry Documents Library.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "DocVQAtest", "organization_count": 1, "organizations": ["llm_stats"], "rank": 353, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/docvqatest?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-doubao-multi-turn-bench", "caveat": "Doubao Multi-Turn Bench evaluates models on multi-turn conversational tasks, measuring context retention, instruction following, and coherent reasoning across extended dialogues.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "instruction_following", "name": "Doubao Multi-Turn Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 354, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/doubao-multi-turn-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1731-dove", "caveat": "DOVE (Dataset Of Variation Evaluation) is a large-scale dataset containing prompt perturbations of various evaluation benchmarks. DOVE（变异评估数据集），这是一个大规模数据集，包含了各种评估基准的提示扰动。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DOVE"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "DOVE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 355, "released": "2025-03-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DOVE"}, {"aliases": [], "benchmark_id": "draco", "caveat": "Deep-research style long-context agent tasks; the retrieval and tool stack is part of the score.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "DRACO", "organization_count": 1, "organizations": ["Tencent"], "rank": 356, "released": null, "source": "model_reports", "url": "https://huggingface.co/datasets/perplexity-ai/draco"}, {"aliases": [], "benchmark_id": "llm-stats-draco", "caveat": "DRACO is a deep research benchmark that evaluates an agent's ability to gather, synthesize, and reason over information to answer complex research questions. Scores are based on official rubrics per question, with the final score being the average across all questions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "DRACO", "organization_count": 1, "organizations": ["llm_stats"], "rank": 357, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/draco?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2045-dragon", "caveat": "DRAGON 是一个用于评估检索增强生成（RAG）系统在俄语新闻语境中事实性与检索能力的动态基准，支持对检索器与生成器组件的全面评估。 DRAGON is a dynamic benchmark for evaluating Retrieval-Augmented Generation (RAG) systems in Russian news contexts, supporting comprehensive assessment of both retriever and generator components.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DRAGON"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "DRAGON", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 358, "released": "2025-07-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DRAGON"}, {"aliases": [], "benchmark_id": "llm-stats-drop", "caveat": "DROP (Discrete Reasoning Over Paragraphs) is a reading comprehension benchmark requiring discrete reasoning over paragraph content. It contains crowdsourced, adversarially-created questions that require resolving references and performing discrete operations like addition, counting, or sorting, demanding comprehensive paragraph understanding beyond paraphrase-and-entity-typing shortcuts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "DROP", "organization_count": 1, "organizations": ["llm_stats"], "rank": 359, "released": "2019-03-01", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/drop?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-536-drop", "caveat": "DROP is a QA dataset which tests comprehensive understanding of paragraphs. In this crowdsourced, adversarially-created, 96k question-answering benchmark, a system must resolve multiple references in a question, map them onto a paragraph, and perform discrete operations over them (such as addition, counting, or sorting). DROP 是一个测试段落综合理解能力的 QA 数据集。在这个众包、对抗性创建的 96K 问题解答基准中，系统必须解析问题中的多个引用，将它们映射到段落中，并对它们执行离散操作（如加法、计数或排序）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DROP"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "DROP", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 360, "released": "2019-03-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DROP"}, {"aliases": [], "benchmark_id": "opencompass-544-ds-1000", "caveat": "DS-1000 is a code generation benchmark with a thousand data science questions spanning seven Python libraries that (1) reflects diverse, realistic, and practical use cases, (2) has a reliable metric, (3) defends against memorization by perturbing questions. DS-1000 是一个代码生成基准测试，包含一千个数据科学问题，涵盖七个Python库，其特点是（1）反映多样化、现实且实用的用例，（2）具有可靠的度量标准，（3）通过扰乱问题来防止记忆化。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DS-1000"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "DS-1000", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 361, "released": "2022-11-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DS-1000"}, {"aliases": [], "benchmark_id": "llm-stats-ds-arena-code", "caveat": "Data Science Arena Code benchmark for evaluating LLMs on realistic data science code generation tasks. Tests capabilities in complex data processing, analysis, and programming across popular Python libraries used in data science workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "DS-Arena-Code", "organization_count": 1, "organizations": ["llm_stats"], "rank": 362, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-arena-code?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ds-fim-eval", "caveat": "DeepSeek's internal Fill-in-the-Middle evaluation dataset for measuring code completion performance improvements in data science contexts", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "DS-FIM-Eval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 363, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ds-fim-eval?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-dsbench-fullstack", "caveat": "DSBench-FullStack is DeepSeek's internal full-stack development test set for evaluating coding agents on end-to-end software engineering tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "DSBench-FullStack", "organization_count": 1, "organizations": ["llm_stats"], "rank": 364, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-fullstack?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-dsbench-hard", "caveat": "DSBench-Hard is DeepSeek's internal test set of difficult coding-agent problems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "DSBench-Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 365, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/dsbench-hard?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-dude", "caveat": "DUDE (Document Understanding Dataset and Evaluation) tests multi-page, multi-domain document understanding and reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "DUDE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 366, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/dude?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1971-dycodeeval", "caveat": "DyCodeEval introduces methods to generate dynamic evaluation dataset and metric. Using multi-agent cooperation to rewrite benchmarks at evaluation time, it generates semantically equivalent, diverse, and non-deterministic problems—reducing data contamination and enabling more trustworthy evaluation. DyCodeEval 提出了一种动态生成评测数据集和评测指标的方法。通过多智能体协作，在评测时对基准题目进行重写，生成语义等价、多样化且非确定性的问题，从而减少数据污染，实现更可信的评测。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DyCodeEval"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "DyCodeEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 367, "released": "2025-06-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DyCodeEval"}, {"aliases": [], "benchmark_id": "llm-stats-dynamath", "caveat": "A multimodal mathematics and reasoning benchmark focused on dynamic visual problem solving.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "DynaMath", "organization_count": 1, "organizations": ["llm_stats"], "rank": 368, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/dynamath?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1374-dynamath", "caveat": "DynaMath is a dynamic visual math benchmark designed for in-depth assessment of VLMs. It includes 501 high-quality, multi-topic seed questions, each represented as a Python program enabling the automatic generation of a much larger set of concrete questions. DynaMath用于评估多模态大模型的数学能力，包括501个高质量、多主题的种子问题，每个问题都以Python程序表示，能够自动生成大量具体的多样化问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/DynaMath"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "DynaMath", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 369, "released": "2024-10-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/DynaMath"}, {"aliases": [], "benchmark_id": "opencompass-1080-e-eval", "caveat": "E-EVAL is the first comprehensive evaluation benchmark specifically tailored for Chinese K-12 education. E-EVAL comprises 4,351 multiple-choice questions spanning primary, middle, and high school levels, covering a diverse array of subjects. E-EVAL 是首个专门针对中国 K-12 教育的综合评估基准。E-EVAL 包含 4,351 道选择题，涵盖小学、初中和高中各个年级，涉及多种学科。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/E-EVAL"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "E-EVAL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 370, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/E-EVAL"}, {"aliases": [], "benchmark_id": "llm-stats-eclektic", "caveat": "A multilingual closed-book question answering dataset that evaluates cross-lingual knowledge transfer in large language models across 12 languages, using knowledge-seeking questions based on Wikipedia articles that exist only in one language", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ECLeKTic", "organization_count": 1, "organizations": ["llm_stats"], "rank": 371, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/eclektic?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1970-editinspector", "caveat": "EditInspector is a novel benchmark for evaluation of text-guided image edits, based on human annotations collected using an extensive template for edit verification. We leverage EditInspector to evaluate the performance of state-of-the-art (SoTA) vision and language models. EditInspector 评估最先进（SoTA）视觉和语言模型在多个维度上评估编辑的性能，包括准确性、瑕疵检测、视觉质量、与图像场景的无缝融合、遵循常识以及描述编辑引起变化的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EditInspector"], "document_share": 0.0008271298593879239, "domain": "创作", "name": "EditInspector", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 372, "released": "2025-06-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EditInspector"}, {"aliases": [], "benchmark_id": "opencompass-1604-egonormia", "caveat": "EgoNormia is a challenging QA benchmark that tests VLMs' ability to reason over norms in context. The datset consists of 1,853 physically grounded egocentric interaction clips from Ego4D and corresponding five-way multiple-choice questions tasks for each. EgoNormia 是一个具有挑战性的问答基准，用于测试 VLMs 在上下文中推理规范的能力。该数据集包含来自 Ego4D 的 1,853 个物理基础化的以自我为中心的交互剪辑，以及每个剪辑对应的五选一多项选择题任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EgoNormia"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "EgoNormia", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 373, "released": "2025-02-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EgoNormia"}, {"aliases": [], "benchmark_id": "llm-stats-egoschema", "caveat": "A diagnostic benchmark for very long-form video language understanding consisting of over 5000 human curated multiple choice questions based on 3-minute video clips from Ego4D, covering a broad range of natural human activities and behaviors", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "EgoSchema", "organization_count": 1, "organizations": ["llm_stats"], "rank": 374, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/egoschema?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1327-ehrnoteqa", "caveat": "EHRNoteQA evaluates LLM's ability to assist clinical decision-making based on electronic health records, comprising 962 different QA pairs each linked to distinct patients' discharge summaries. EHRNoteQA用于评估LLM基于电子健康记录辅助临床决策的能力，由962个问答对组成，每个问答对都与不同患者的出院总结相关联。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EHRNoteQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "EHRNoteQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 375, "released": "2024-02-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EHRNoteQA"}, {"aliases": [], "benchmark_id": "opencompass-2571-elbench", "caveat": "ELBench is a multidimensional benchmark for education-facing large language models. It covers safety and trustworthiness, high-level educational cultivation, and  general capabilities, combining objective evaluation with LLM-as-a-Judge. ELBench 是面向教育场景大语言模型的多维评测集，覆盖安全可信、高阶育人和通用能力。公开任务包含安全回答、教育开放任务、选择题、数学与推理任务，并结合客观规则和大模型判决进行评测。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ELBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "ELBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 376, "released": "2026-06-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ELBench"}, {"aliases": [], "benchmark_id": "opencompass-1243-embodiedagentinterface", "caveat": "Embodied Agent Interface supports the formalization of various types of tasks and input-output specifications of LLM-based modules, offering a comprehensive assessment of LLMs' performance for different subtasks and pinpointing the strengths and weaknesses in LLM-powered embodied AI systems. Embodied Agent Interface支持各种类型的任务和基于LLM模块的输入输出规范的形式化，对LLM在不同子任务中的性能进行了全面评估，指出了基于LLM的具身AI系统的优势和劣势。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EmbodiedAgentInterface"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "EmbodiedAgentInterface", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 377, "released": "2024-06-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedAgentInterface"}, {"aliases": [], "benchmark_id": "opencompass-1564-embodiedbench", "caveat": "EmbodiedBench is a comprehensive benchmark designed to evaluate Multi-modal Large Language Models (MLLMs) as embodied agents. EmbodiedBench，这是一个旨在评估多模态大型语言模型作为具身智能体的全面基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EmbodiedBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "EmbodiedBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 378, "released": "2025-02-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EmbodiedBench"}, {"aliases": [], "benchmark_id": "llm-stats-embspatialbench", "caveat": "EmbSpatialBench evaluates embodied spatial understanding and reasoning capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "EmbSpatialBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 379, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/embspatialbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-emma", "caveat": "EMMA (Enhanced MultiModal reAsoning) is a benchmark for organic multimodal reasoning across mathematics, physics, chemistry, and coding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "EMMA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 380, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/emma?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1640-emma", "caveat": "EMMA is composed of 2,788 problems, of which 1,796 are newly constructed, across four domains. Within each subject, we further provide fine-grained labels for each question based on the specific skills it measures. EMMA 由 2,788 个问题组成，其中 1,796 个是新构建的，涵盖四个领域。在每个主题中，我们根据所测量的具体技能为每个问题提供细粒度标签。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EMMA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "EMMA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 381, "released": "2025-01-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EMMA"}, {"aliases": [], "benchmark_id": "artificial-analysis-enterpriseops-gym-aa", "caveat": "Agentic business operations", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "EnterpriseOps-Gym-AA", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 382, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa"}, {"aliases": [], "benchmark_id": "opencompass-522-eprstmt", "caveat": "(E-commerce Product Review Dataset for Sentiment Analysis), also known as EPRSTMT, is a binary sentiment analysis dataset based on product reviews on e-commerce platform. Each sample is labelled as Positive or Negative. It collect by ICIP Lab of Beijing Normal University. EPRSTMT，也称作电子商务产品评论情感分析数据集，是一个基于电子商务平台上的产品评论的二元情感分析数据集。每个样本都被标记为积极或消极。该数据集由北京师范大学 ICIP 实验室收集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EPRSTMT"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "EPRSTMT", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 383, "released": "2021-07-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EPRSTMT"}, {"aliases": [], "benchmark_id": "llm-stats-eq-bench", "caveat": "EQ-Bench is an LLM-judged test evaluating active emotional intelligence abilities, understanding, insight, empathy, and interpersonal skills. The test set contains 45 challenging roleplay scenarios, most of which constitute pre-written prompts spanning 3 turns. The benchmark evaluates the performance of models by validating responses against several criteria and conducts pairwise comparisons to report a normalized Elo computation for each model.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "EQ-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 384, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/eq-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1892-er-reason", "caveat": "ER-Reason is a large-scale benchmark suite for evaluating the clinical reasoning capabilities of large language models (LLMs) in the emergency room (ER) — a high-stakes environment where clinicians make rapid, life-critical decisions. ER-Reason 是一个大规模基准套件，用于评估大语言模型（LLMs）在急诊室（ER）中的临床推理能力。急诊室是一个高风险环境，临床医生需要快速做出关乎生命的关键决策。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ER-Reason"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "ER-Reason", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 385, "released": "2025-05-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ER-Reason"}, {"aliases": [], "benchmark_id": "llm-stats-erqa", "caveat": "Embodied Reasoning Question Answering benchmark consisting of 400 multiple-choice visual questions across spatial reasoning, trajectory reasoning, action reasoning, state estimation, and multi-view reasoning for evaluating AI capabilities in physical world interactions", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ERQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 386, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/erqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-evalplus", "caveat": "A rigorous code synthesis evaluation framework that augments existing datasets with extensive test cases generated by LLM and mutation-based strategies to better assess functional correctness of generated code, including HumanEval+ with 80x more test cases", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "EvalPlus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 387, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/evalplus?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1843-ewmbench", "caveat": "EWMBench is a benchmark for evaluating Embodied World Models, covering aspects such as scene consistency, motion correctness, and semantic alignment. It includes a diverse dataset and multi-dimensional metrics tailored for embodied manipulation tasks. EWMBench 是一个用于评估具身世界模型的基准，涵盖“场景一致性”、“动作正确性”和“语义对齐”等方面。它包含多样化的数据集和面向具身任务的多维评估指标，能够揭示现有模型的局限，并为生成具物理基础、任务导向的视频提供评价标准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/EWMBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "EWMBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 388, "released": "2025-05-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/EWMBench"}, {"aliases": [], "benchmark_id": "llm-stats-exploitbench", "caveat": "ExploitBench is a cybersecurity benchmark that evaluates a model's ability to discover and exploit software vulnerabilities, reported as the fraction of challenges where the model captures the target (Cap%).", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "ExploitBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 389, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-exploitgym", "caveat": "ExploitGym is a large-scale, realistic benchmark built from real-world vulnerabilities across userspace programs, Google's V8 engine, and the Linux kernel. Given a vulnerability and a proof-of-vulnerability input, agents must craft a working end-to-end exploit that achieves unauthorized code execution.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "ExploitGym", "organization_count": 1, "organizations": ["llm_stats"], "rank": 390, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/exploitgym?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-facts-grounding", "caveat": "A benchmark evaluating language models' ability to generate factually accurate and well-grounded responses based on long-form input context, comprising 1,719 examples with documents up to 32k tokens requiring detailed responses that are fully grounded in provided documents", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FACTS Grounding", "organization_count": 1, "organizations": ["llm_stats"], "rank": 391, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/facts-grounding?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-factscore", "caveat": "A fine-grained atomic evaluation metric for factual precision in long-form text generation that breaks generated text into atomic facts and computes the percentage supported by reliable knowledge sources, with automated assessment using retrieval and language models", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FActScore", "organization_count": 1, "organizations": ["llm_stats"], "rank": 392, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/factscore?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1980-falsereject", "caveat": "Safety alignment approaches in large language models (LLMs) often lead\nto the over-refusal of benign queries, significantly diminishing their utility\nin sensitive scenarios. To address this challenge, we introduce FalseReject,\na comprehensive resource containing 16k seemingly toxic queries accompani FalseReject 构建了包含 16 000 条表面“有毒”但实为良性的查询样本，覆盖 44 个安全相关类别，并提出一种基于图信息的对抗多智能体交互框架，用以生成多样且复杂的提示–响应对，并在响应中引入显式推理链，帮助模型更准确地区分安全与不安全上下文；该工作还为标准指令调优模型和推理导向模型分别准备了专项训练集，并附带人工标注的基准测试集，针对 29 款最先进 LLM 进行了大规模评估，结果表明经 FalseReject 监督微调后，模型在显著减少对良性查询的过度拒绝的同时，不仅未损失整体安全性，也保持了语言生成能力，为敏感场景下提升 LLM 可用性提供了首个系统化、可复现的资源与方法框", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FalseReject"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "FalseReject", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 393, "released": "2025-05-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/FalseReject"}, {"aliases": [], "benchmark_id": "opencompass-1733-feabench", "caveat": "FEABench is a benchmark to evaluate the ability of large language models (LLMs) and LLM agents to simulate and solve physics, mathematics and engineering problems using finite element analysis (FEA). FEABench，一个用于评估大型语言模型和LLM代理使用有限元分析（FEA）模拟和解决物理、数学及工程问题能力的基准测试。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FEABench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "FEABench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 394, "released": "2025-04-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/FEABench"}, {"aliases": [], "benchmark_id": "opencompass-1618-fedmabench", "caveat": "FedMABench is an open-source benchmark for federated training and evaluation of mobile agents, specifically designed for heterogeneous scenarios. FedMABench 是一个开源的联邦训练和评估移动代理的基准，特别为异构场景设计。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FedMABench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "FedMABench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 395, "released": "2025-03-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/FedMABench"}, {"aliases": [], "benchmark_id": "llm-stats-figqa", "caveat": "FigQA is a multiple-choice benchmark on interpreting scientific figures from biology papers. It evaluates dual-use biological knowledge and multimodal reasoning relevant to bioweapons development.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "FigQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 396, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/figqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-895-fin-eva", "caveat": "Ant Group and Shanghai University of Finance and Economics jointly launched the financial benchmark，Fin-Eva Version 1.0, covering multiple financial scenarios and subjects such as wealth management, insurance, investment research. The number of this benchmark's questions reaches 13,000+. 蚂蚁集团、上海财经大学联合推出金融评测集Fin-Eva Version 1.0，覆盖财富管理、保险、投资研究等多个金融场景以及金融专业主题学科，总评测题数目达到13,000+。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Fin-Eva"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "Fin-Eva", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 397, "released": "2023-12-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Fin-Eva"}, {"aliases": [], "benchmark_id": "llm-stats-finance-agent", "caveat": "Finance Agent is a benchmark for evaluating AI models on agentic financial analysis tasks, testing their ability to process financial data, perform calculations, and generate accurate analyses across various financial domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Finance Agent", "organization_count": 1, "organizations": ["llm_stats"], "rank": 398, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-finance-agent-v1-1", "caveat": "Finance Agent v1.1 is an agentic financial-analysis benchmark that evaluates models on real-world finance workflows, including retrieving and reasoning over financial documents and performing multi-step calculations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Finance Agent v1.1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 399, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v1.1?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-finance-agent-v2", "caveat": "Finance Agent v2 is an agentic financial-analysis benchmark from Vals that evaluates models on real-world finance workflows, measuring their ability to retrieve and reason over financial documents, perform multi-step calculations, and produce accurate analyses.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Finance Agent v2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 400, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/finance-agent-v2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-finqa", "caveat": "A large-scale dataset for numerical reasoning over financial data with question-answering pairs written by financial experts, featuring complex numerical reasoning and understanding of heterogeneous representations with annotated gold reasoning programs for full explainability", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "FinQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 401, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/finqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-finsearchcomp-t2-t3", "caveat": "FinSearchComp T2&T3 is a combined benchmark for evaluating financial search and reasoning capabilities on Tier 2 and Tier 3 tasks, testing models' ability to retrieve and analyze complex financial information using tools.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FinSearchComp T2&T3", "organization_count": 1, "organizations": ["llm_stats"], "rank": 402, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t2-t3?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-finsearchcomp-t3", "caveat": "FinSearchComp-T3 is a benchmark for evaluating financial search and reasoning capabilities, testing models' ability to retrieve and analyze financial information using tools.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FinSearchComp-T3", "organization_count": 1, "organizations": ["llm_stats"], "rank": 403, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/finsearchcomp-t3?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-flame-vlm-code", "caveat": "Flame-VLM-Code evaluates multimodal models on visual code generation tasks, measuring ability to generate code from visual inputs such as UI mockups and design specifications.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Flame-VLM-Code", "organization_count": 1, "organizations": ["llm_stats"], "rank": 404, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/flame-vlm-code?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-945-flames", "caveat": "Flames is a highly adversarial benchmark in Chinese for LLM's value alignment evaluation developed by Shanghai AI Lab and Fudan NLP Group. Flames meticulously designs a dataset of 2,251 highly adversarial, manually crafted prompts, each tailored to probe a specific value dimension (i.e., Fairness, Safety, Morality, Legality, Data protection). Currently,  Flames releases 1,000 prompts for public use (Flames_1k_Chinese). Flames 是上海人工智能实验室和复旦大学 NLP团队开发的 LLM 价值对齐方向的中文高度对抗性基准。Flames 精心设计了一个由 2,251 个高度对抗性、人工创建的提示词成的评测集，每个提示词都经过精心设计，以探究特定的价值维度（即公平、安全、道德、合法、数据保护）。目前，Flames 发布了 1,000 个提示词供公众使用（Flames_1k_Chinese）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Flames"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "Flames", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 405, "released": "2024-03-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Flames"}, {"aliases": [], "benchmark_id": "llm-stats-flenqa", "caveat": "Flexible Length Question Answering dataset for evaluating the impact of input length on reasoning performance of language models, featuring True/False questions embedded in contexts of varying lengths (250-3000 tokens) across three reasoning tasks: Monotone Relations, People In Rooms, and simplified Ruletaker", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "FlenQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 406, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/flenqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-fleurs", "caveat": "Few-shot Learning Evaluation of Universal Representations of Speech - a parallel speech dataset in 102 languages built on FLoRes-101 with approximately 12 hours of speech supervision per language for tasks including ASR, speech language identification, translation and retrieval. Scores are shown as speech recognition accuracy (1 - word error rate), so higher is better.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500"], "document_share": 0.0008271298593879239, "domain": "speech_to_text", "name": "FLEURS", "organization_count": 1, "organizations": ["llm_stats"], "rank": 407, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/fleurs?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-509-flores", "caveat": "Flores is a benchmark dataset for machine translation between English and low-resource languages, which consists of sentences translated from Wikipedia, involving English and four low-resource languages, namely Nepali, Sinhala, Khmer and Pashto. Flores has two versions, we use the Flores-101 version here, which is the second version including 101 languages besides english. Flores是一个用于评估低资源语言机器翻译的基准数据集，它包含了从维基百科翻译的句子，涉及英语和四种低资源语言，分别是尼泊尔语、僧伽罗语、高棉语和普什图语。Flores有两个版本，我们这里使用的是第一个版本Flores-101，它包含有除英语外的101种语言。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Flores"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "Flores", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 408, "released": "2021-06-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Flores"}, {"aliases": [], "benchmark_id": "opencompass-1324-flub", "caveat": "FLUB evaluates the reasoning and understanding abilities of LLMs. It includes three tasks with increasing difficulty, consisting of the tricky, humorous, and misleading texts collected from the real internet environment. FLUB用于评估LLM的推理和理解能力，其中包含3个难度递进的任务，由从真实互联网环境中收集的狡猾、幽默和误导性的文本构成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FLUB"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "FLUB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 409, "released": "2024-02-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/FLUB"}, {"aliases": [], "benchmark_id": "opencompass-1671-forensics-bench", "caveat": "A Comprehensive Forgery Detection Benchmark Suite for Large Vision Language Models A Comprehensive Forgery Detection Benchmark Suite for Large Vision Language Models", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Forensics-bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "Forensics-bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 410, "released": "2025-03-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Forensics-bench"}, {"aliases": [], "benchmark_id": "opencompass-1737-fortisavqa", "caveat": "FortisAVQA is the first dataset designed to assess the robustness of AVQA models. FortisAVQA，这是首个用于评估AVQA模型鲁棒性的数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FortisAVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "FortisAVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 411, "released": "2025-04-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/FortisAVQA"}, {"aliases": [], "benchmark_id": "llm-stats-frames", "caveat": "Factuality, Retrieval, And reasoning MEasurement Set - a unified evaluation dataset of 824 challenging multi-hop questions for testing retrieval-augmented generation systems across factuality, retrieval accuracy, and reasoning capabilities, requiring integration of 2-15 Wikipedia articles per question", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FRAMES", "organization_count": 1, "organizations": ["llm_stats"], "rank": 412, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frames?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1145-freb-tqa", "caveat": "FREB-TQA is a Fine-grained Robustness Evaluation Benchmark for Table Question Answering. FREB-TQA 是一个细粒度的稳健性评估基准，专注于表格问答（TQA）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/FREB-TQA"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "FREB-TQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 413, "released": "2024-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/FREB-TQA"}, {"aliases": [], "benchmark_id": "llm-stats-french-mmlu", "caveat": "French version of MMLU-Pro, a multilingual benchmark for evaluating language models' cross-lingual reasoning capabilities across 14 diverse domains including mathematics, physics, chemistry, law, engineering, psychology, and health.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "French MMLU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 414, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/french-mmlu?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontier-science", "caveat": "Frontier Science is a benchmark of exceptionally challenging scientific reasoning problems spanning advanced natural-science domains, designed to test expert-level scientific understanding and multi-step reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Frontier Science", "organization_count": 1, "organizations": ["llm_stats"], "rank": 415, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-science?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontier-bench-v0-1", "caveat": "Frontier-Bench v0.1 evaluates agentic terminal coding. Anthropic reports results using the mini-SWE-agent harness and a GKE backend, measured as mean reward across five attempts per task.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Frontier-Bench v0.1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 416, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-bench-v0.1?top_n=500"}, {"aliases": ["FrontierChallenge", "Frontier Challenge"], "benchmark_id": "frontier_challenge", "caveat": "Pass-rate results are model-scaffold measurements over the 97 released tasks with one trajectory per system-task pair, so a reported percentage is not a model-only number and does not estimate reliability beyond this release or under a different scaffold.", "document_count": 1, "document_ids": ["model_reports:frontier_challenge_leaderboard_2026_08_25"], "document_share": 0.0008271298593879239, "domain": "scientific_agent", "name": "FrontierChallenge", "organization_count": 1, "organizations": ["ApodexAI"], "rank": 417, "released": "2026-08-25", "source": "model_reports", "url": "https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/"}, {"aliases": ["FrontierCode", "FrontierCode (Diamond)"], "benchmark_id": "frontiercode", "caveat": "Cognition's production-standard coding evaluation, asking whether models write *good* code rather than merely correct code. Unrelated to Terminal-Bench's Frontier-Bench despite the shared prefix. Reported per reasoning-effort setting (the Diamond subset at xhigh), so the effort level has to travel with the number. First-party to a competing coding agent's vendor, and revised as 1.1 on 2026-07-07, so the methodology version has to travel with the number too.", "document_count": 1, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5"], "document_share": 0.0008271298593879239, "domain": "coding_agent", "name": "FrontierCode", "organization_count": 1, "organizations": ["Anthropic"], "rank": 418, "released": "2026-06-08", "source": "model_reports", "url": "https://cognition.com/blog/frontier-code"}, {"aliases": [], "benchmark_id": "llm-stats-frontiercode", "caveat": "FrontierCode is Cognition's coding evaluation that tests whether models can pass difficult coding tasks while meeting the standards of high-quality production codebases. The Diamond subset contains the hardest problems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FrontierCode", "organization_count": 1, "organizations": ["llm_stats"], "rank": 419, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontiercode-1-1", "caveat": "FrontierCode 1.1 evaluates whether coding-agent changes are mergeable, using unit tests, maintainer-defined rubrics, and verifiers. Runs flagged for unfair internet use receive a zero score.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FrontierCode 1.1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 420, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercode-1.1?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontiercs", "caveat": "FrontierCS is a benchmark of frontier computer-science problems requiring deep theoretical understanding and rigorous multi-step reasoning at the edge of the field.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FrontierCS", "organization_count": 1, "organizations": ["llm_stats"], "rank": 421, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiercs?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontiermath", "caveat": "A benchmark of hundreds of original, exceptionally challenging mathematics problems crafted and vetted by expert mathematicians, covering most major branches of modern mathematics from number theory and real analysis to algebraic geometry and category theory.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "FrontierMath", "organization_count": 1, "organizations": ["llm_stats"], "rank": 422, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontiermath-tier-4-v2", "caveat": "FrontierMath Tier 4 subset from the v2 evaluation release.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "FrontierMath Tier 4 (v2)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 423, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontiermath-tier-4-v2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontierscience-olympiad", "caveat": "FrontierScience Olympiad is a benchmark of olympiad-level scientific reasoning problems, testing expert understanding and multi-step reasoning across advanced natural-science domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "FrontierScience Olympiad", "organization_count": 1, "organizations": ["llm_stats"], "rank": 424, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-olympiad?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontierscience-research", "caveat": "FrontierScience Research is a benchmark evaluating AI models on cutting-edge scientific research questions requiring deep domain expertise, multi-step reasoning, and synthesis of complex scientific concepts across disciplines.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FrontierScience Research", "organization_count": 1, "organizations": ["llm_stats"], "rank": 425, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierscience-research?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontierswe", "caveat": "FrontierSWE measures whether an agent can complete open-ended technical projects at the scale of hours to tens of hours, spanning systems optimization, large-scale code construction, and applied ML research. Performance is reported as a dominance score, where higher is better.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "FrontierSWE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 426, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontierswe?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-frontier-swe-impl", "caveat": "FrontierSWE (Impl.) evaluates software engineering implementation ability and reports model ranking on implementation tasks. Lower rank is better.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "FrontierSWE (Impl.)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 427, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/frontier-swe-impl?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-fullstackbench-en", "caveat": "English subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FullStackBench en", "organization_count": 1, "organizations": ["llm_stats"], "rank": 428, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-en?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-fullstackbench-zh", "caveat": "Chinese subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "FullStackBench zh", "organization_count": 1, "organizations": ["llm_stats"], "rank": 429, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/fullstackbench-zh?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-functionalmath", "caveat": "A functional variant of the MATH benchmark that tests language models' ability to generalize reasoning patterns across different problem instances, revealing the reasoning gap between static and functional performance.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "FunctionalMATH", "organization_count": 1, "organizations": ["llm_stats"], "rank": 430, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/functionalmath?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1599-gaia", "caveat": "GAIA is a benchmark which aims at evaluating next-generation LLMs (LLMs with augmented capabilities due to added tooling, efficient prompting, access to search, etc), mading of more than 450 non-trivial question with an unambiguous answer, requiring different levels of tooling and autonomy to solve. GAIA 是一个旨在评估下一代LLMs（由于增加了工具、高效的提示、访问搜索等功能而具有增强能力的LLMs）的基准，由超过 450 个非平凡问题组成，这些问题有明确的答案，需要不同层次的工具和自主性来解决。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAIA"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "GAIA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 431, "released": "2023-11-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GAIA"}, {"aliases": [], "benchmark_id": "llm-stats-gaia2", "caveat": "GAIA2 evaluates general-purpose AI agents on real-world, multi-step questions that require reasoning, tool use, and information retrieval.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "GAIA2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 432, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gaia2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-gameworld", "caveat": "GameWorld evaluates agents on interactive game environments, testing perception, planning, and sequential decision-making to accomplish in-game objectives.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "GameWorld", "organization_count": 1, "organizations": ["llm_stats"], "rank": 433, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gameworld?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-500-gaokao-bench", "caveat": "GAOKAO-bench is an evaluation framework that utilizes Chinese high school entrance examination (GAOKAO) questions as a dataset to evaluate the language understanding and logical reasoning abilities of large language models. GAOKAO-bench是一个以中国高考题目为数据集，旨在提供和人类对齐的，直观，高效地测评大模型语言理解能力、逻辑推理能力的测评框架", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAOKAO-Bench"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "GAOKAO-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 434, "released": "2023-05-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1083-gaokao-mm", "caveat": "GAOKAO-MM is a multimodal benchmark based on the Chinese College Entrance Examination (GAOKAO), comprising of 8 subjects and 12 types of images, such as diagrams, function graphs, maps and photos. GAOKAO-MM 是一个基于中国高考的多模态基准，包含 8 个科目和 12 种图像类型，例如图表、函数图、地图和照片。GAOKAO-MM 源自本土中文语境，并对模型的能力设置了人类水平的要求，包括感知、理解、知识和推理。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAOKAO-MM"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "GAOKAO-MM", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 435, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GAOKAO-MM"}, {"aliases": [], "benchmark_id": "opencompass-2574-gauge", "caveat": "GAUGE: A Physical‑Realism Evaluation Benchmark Leveraging real‑world replicated experimental data as ground‑truth references, it delivers systematic evaluation for world models and physics engines across rigid bodies, ropes, fabrics, and 3D soft bodies. 物理真实度评测基准GAUGE，以真实重复实验数据为对照标准，为世界模型与物理引擎提供跨刚体、绳索、织物和三维软体的系统化评测。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GAUGE"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "GAUGE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 436, "released": "2026-08-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GAUGE"}, {"aliases": ["GDP.pdf"], "benchmark_id": "gdp_pdf", "caveat": "Vision-based knowledge work over 100 PDFs across 10 domains, reported no-tools. Published by Surge AI with a public dataset (huggingface.co/datasets/surgeai/GDP.pdf), harness (github.com/surge-ai/gdp-pdf) and paper (arXiv:2607.11192). The name resembles GDPval but the two are separate instruments and must not be merged.", "document_count": 1, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "GDP.pdf", "organization_count": 1, "organizations": ["Anthropic"], "rank": 437, "released": "2026-04-14", "source": "model_reports", "url": "https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world"}, {"aliases": [], "benchmark_id": "llm-stats-gdp-pdf", "caveat": "GDP.pdf is a knowledge-work vision benchmark that evaluates models on economically valuable professional tasks presented as visual documents (PDFs), testing document-based reasoning, chart and table interpretation, and problem solving without tools.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "GDP.pdf", "organization_count": 1, "organizations": ["llm_stats"], "rank": 438, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdp-pdf?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-gdpval", "caveat": "GDPval is an OpenAI benchmark evaluating AI models on economically valuable, real-world knowledge-work tasks spanning many professional occupations and industries.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "GDPval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 439, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-gdpval-aa", "caveat": "GDPval-AA evaluates AI agents on economically valuable professional knowledge-work tasks and reports performance as an Elo score.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "GDPval-AA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 440, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-aa?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-gdpval-aa-v2-raw-elo", "caveat": "Agentic real-world work tasks", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/gdpval-aa"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "GDPval-AA v2 (Elo)", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 441, "released": "2026-06-15", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}, {"aliases": [], "benchmark_id": "artificial-analysis-gdpval-aa-v2-normalized-score", "caveat": "Agentic real-world work tasks", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/gdpval-aa"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "GDPval-AA v2 (normalized score)", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 442, "released": "2026-06-15", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/gdpval-aa"}, {"aliases": [], "benchmark_id": "llm-stats-gdpval-mm", "caveat": "GDPval-MM is the multimodal variant of the GDPval benchmark, evaluating AI model performance on real-world economically valuable tasks that require processing and generating multimodal content including documents, slides, diagrams, spreadsheets, images, and other professional deliverables across diverse industries.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "GDPval-MM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 443, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-mm?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-gdpval-rubrics", "caveat": "GDPval-Rubrics evaluates AI model performance on economically valuable knowledge work tasks drawn from the public GDPval dataset. It uses pointwise scoring based on public rubrics, with the environment aligned to the GDPval-AA scaffolding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "GDPval-Rubrics", "organization_count": 1, "organizations": ["llm_stats"], "rank": 444, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gdpval-rubrics?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1376-genai-bench", "caveat": "GenAI-Bench is a benchmark designed to benchmark MLLMs’s ability in judging the quality of AI generative contents, containing over 40,000 human ratings to evaluate the performance of MLLMs on aligning with human preferences. GenAI-Bench用于衡量MLLM判断AI生成内容质量的能力，包含超过40000个人工评分用以评估大模型与人类偏好的一致性。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GenAI-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "GenAI-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 445, "released": "2024-06-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GenAI-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-genebench", "caveat": "GeneBench is an evaluation focused on multi-stage scientific data analysis in genetics and quantitative biology. Tasks require reasoning about ambiguous or noisy data with minimal supervisory guidance, addressing realistic obstacles such as hidden confounders or QC failures, and correctly implementing and interpreting modern statistical methods.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "GeneBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 446, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-genebench-pro", "caveat": "GeneBench-Pro is a research-level benchmark of 129 multi-stage computational-biology problems spanning genomics, quantitative biology, and translational biomedicine. Each problem gives the agent a messy dataset, brief context, and a target estimand, and requires navigating dependent inferential decision points to reach a verifiable answer.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "GeneBench-Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 447, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/genebench-pro?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2081-general-bench", "caveat": "General Bench is a set of universal evaluation benchmarks for multimodal large models, covering language, image, video, audio, and 3D five modalities, with a total of 145 skills, over 700 tasks, and 325800 samples. General-Bench 是一套面向多模态大模型的通用评测基准，涵盖语言、图像、视频、音频和 3D 五大模态，共计 145 项技能、700 余个任务，包含 325 800 条样本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/General-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "General-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 448, "released": "2025-05-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/General-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-giantsteps-tempo", "caveat": "A dataset for tempo estimation in electronic dance music containing 664 2-minute audio previews from Beatport, annotated from user corrections for evaluating automatic tempo estimation algorithms.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500"], "document_share": 0.0008271298593879239, "domain": "audio", "name": "GiantSteps Tempo", "organization_count": 1, "organizations": ["llm_stats"], "rank": 449, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/giantsteps-tempo?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1865-gitgoodbench", "caveat": "GitGoodBench Lite is a subset of 900 samples for evaluating the performance of AI agents in resolving git tasks (see Supported Scenarios). GitGoodBench Lite是一个包含900个样本的子集，用于评估AI智能体在解决git任务方面的性能（参见支持的场景）。数据集中的样本在编程语言Python、Java和Kotlin以及样本类型合并冲突解决和文件提交语法之间均匀分布。因此，该数据集包含每种样本类型和编程语言各150个样本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GitGoodBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "GitGoodBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 450, "released": "2025-05-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GitGoodBench"}, {"aliases": [], "benchmark_id": "llm-stats-global-piqa", "caveat": "Global PIQA is a multilingual commonsense reasoning benchmark that evaluates physical interaction knowledge across 100 languages and cultures. It tests AI systems' understanding of physical world knowledge in diverse cultural contexts through multiple choice questions about everyday situations requiring physical commonsense.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "physics", "name": "Global PIQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 451, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/global-piqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-global-mmlu", "caveat": "A comprehensive multilingual benchmark covering 42 languages that addresses cultural and linguistic biases in evaluation, with improved translation quality and culturally sensitive question subsets.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Global-MMLU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 452, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-global-mmlu-lite", "caveat": "A lightweight version of Global MMLU benchmark that evaluates language models across multiple languages while addressing cultural and linguistic biases in multilingual evaluation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Global-MMLU-Lite", "organization_count": 1, "organizations": ["llm_stats"], "rank": 453, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/global-mmlu-lite?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1927-gmaimmbench", "caveat": "GMAI-MMBench is the most comprehensive and structured benchmark developed to evaluate Large Vision-Language Models (LVLMs) in general medical artificial intelligence (GMAI) applications. It addresses the limitations of existing benchmarks that typically focus on narrow domains and lack perceptual di GMAI-MMBench 是目前最全面、结构化最完善的通用医学人工智能（GMAI）基准数据集，专为评估大规模视觉语言模型（LVLMs）在医疗领域中的表现而设计。针对现有医学基准通常局限于特定学科、感知粒度单一的问题，GMAI-MMBench 从285个真实医学数据集构建而成，涵盖39种医学影像模态、18个临床任务、18个医学科室，以及4种不同的感知粒度。所有任务采用视觉问答（VQA）形式组织，具备良好的交互性和通用性。其独特的词汇树结构允许用户针对具体研究需求灵活定制评估路径，支持多样化的评测场景。我们对50个主流LVLMs进行了系统测试，发现即使是目前最先进的 GPT-4o，其准确率也仅为5", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GMAIMMBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "GMAIMMBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 454, "released": "2024-08-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GMAIMMBench"}, {"aliases": [], "benchmark_id": "opencompass-1128-gorilla", "caveat": "Gorilla enables LLMs to use tools by invoking APIs. Given a natural language query, Gorilla comes up with the semantically- and syntactically- correct API to invoke. Gorilla 使大语言模型能够通过调用 API 使用工具。针对自然语言查询，Gorilla 能够生成语义和语法上正确的 API 调用。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Gorilla"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "Gorilla", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 455, "released": "2023-05-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Gorilla"}, {"aliases": [], "benchmark_id": "llm-stats-gorilla-benchmark-api-bench", "caveat": "APIBench, a comprehensive dataset of over 11,000 instruction-API pairs from HuggingFace, TorchHub, and TensorHub APIs for evaluating language models' ability to generate accurate API calls.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Gorilla Benchmark API Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 456, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gorilla-benchmark-api-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-govreport", "caveat": "A long document summarization dataset consisting of reports from government research agencies including Congressional Research Service and U.S. Government Accountability Office, with significantly longer documents and summaries than other datasets.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "GovReport", "organization_count": 1, "organizations": ["llm_stats"], "rank": 457, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/govreport?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-gpqa", "caveat": "A challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. Questions are Google-proof and extremely difficult, with PhD experts reaching 65% accuracy.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "physics", "name": "GPQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 458, "released": "2023-11-20", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1135-gpqa", "caveat": "GPQA, a challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. GPQA 包含 448 道由生物学、物理学和化学领域专家撰写的多项选择题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GPQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "GPQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 459, "released": "2023-11-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GPQA"}, {"aliases": [], "benchmark_id": "llm-stats-gpqa-biology", "caveat": "Biology subset of GPQA, containing challenging multiple-choice questions written by domain experts in biology. These Google-proof questions require graduate-level knowledge and reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "GPQA Biology", "organization_count": 1, "organizations": ["llm_stats"], "rank": 460, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-biology?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-gpqa-chemistry", "caveat": "Chemistry subset of GPQA, containing challenging multiple-choice questions written by domain experts in chemistry. These Google-proof questions require graduate-level knowledge and reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "GPQA Chemistry", "organization_count": 1, "organizations": ["llm_stats"], "rank": 461, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-chemistry?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-gpqa-diamond", "caveat": "Scientific reasoning", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/gpqa-diamond"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "GPQA Diamond", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 462, "released": "2023-11-20", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/gpqa-diamond"}, {"aliases": [], "benchmark_id": "llm-stats-gpqa-physics", "caveat": "Physics subset of GPQA, containing challenging multiple-choice questions written by domain experts in physics. These Google-proof questions require graduate-level knowledge and reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500"], "document_share": 0.0008271298593879239, "domain": "physics", "name": "GPQA Physics", "organization_count": 1, "organizations": ["llm_stats"], "rank": 463, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gpqa-physics?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1740-gpt-imgeval", "caveat": "GPT-ImgEval, quantitatively and qualitatively diagnoses GPT-4o's performance across three critical dimensions: (1) generation quality, (2) editing proficiency, and (3) world knowledge-informed semantic synthesis. GPT-ImgEval，从三个关键维度对GPT-4o的性能进行定量和定性诊断：（1）生成质量，（2）编辑能力，以及（3）基于世界知识的语义合成能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GPT-ImgEval"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "GPT-ImgEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 464, "released": "2025-04-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GPT-ImgEval"}, {"aliases": [], "benchmark_id": "opencompass-1098-grailqa", "caveat": "GrailQA is a new large-scale, high-quality dataset for question answering on knowledge bases (KBQA) on Freebase with 64,331 questions annotated with both answers and corresponding logical forms in different syntax (i.e., SPARQL, S-expression, etc.). GrailQA 是一个大规模高质量数据集，用于知识库问答，包含 64,331 个问题，并附有答案和不同语法的相应逻辑形式。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GrailQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "GrailQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 465, "released": "2021-02-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GrailQA"}, {"aliases": [], "benchmark_id": "llm-stats-graphwalks", "caveat": "GraphWalks is a synthetic multi-hop long-context reasoning benchmark in which a model is given an edge-list representation of a graph and must traverse it to find neighboring nodes (via breadth-first search) or parent nodes for a given start node. Performance is reported as F1 of the model-predicted answer set versus the ground truth.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "GraphWalks", "organization_count": 1, "organizations": ["llm_stats"], "rank": 466, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-graphwalks-bfs-1m", "caveat": "GraphWalks BFS variant evaluated on 1M-token contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "Graphwalks BFS 1M", "organization_count": 1, "organizations": ["llm_stats"], "rank": 467, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-1m?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-graphwalks-bfs-128k", "caveat": "A graph reasoning benchmark that evaluates language models' ability to perform breadth-first search (BFS) operations on graphs with context length under 128k tokens, returning nodes reachable at specified depths.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Graphwalks BFS <128k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 468, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3C128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-graphwalks-bfs-128k-2", "caveat": "A graph reasoning benchmark that evaluates language models' ability to perform breadth-first search (BFS) operations on graphs with context length over 128k tokens, testing long-context reasoning capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "Graphwalks BFS >128k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 469, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-bfs-%3E128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-graphwalks-parents-128k", "caveat": "A graph reasoning benchmark that evaluates language models' ability to find parent nodes in graphs with context length under 128k tokens, requiring understanding of graph structure and edge relationships.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Graphwalks parents <128k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 470, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3C128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-graphwalks-parents-128k-2", "caveat": "A graph reasoning benchmark that evaluates language models' ability to find parent nodes in graphs with context length over 128k tokens, testing long-context reasoning and graph structure understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "Graphwalks parents >128k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 471, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/graphwalks-parents-%3E128k?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2027-groundingsuite", "caveat": "GroundingSuite is designed to test the localization capabilities of multimodal models. It created 3,720 pixel-level data entries based on COCO Unlabeled images. This dataset covers four dimensions: Stuff Class Object, Multi-Object, Part-Level Object, and Single Object. GroundingSuite 用来测试多模态模型的定位能力。它通过半自动标注和人工筛选在COCO Unlabel的图片基础上创建了3720条pixel-level的数据，覆盖Stuff Class Object, Multi Object, Part Level Object, Single Object四个维度，是一个全面评测多模态模型定位能力的测试集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GroundingSuite"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "GroundingSuite", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 472, "released": "2025-07-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GroundingSuite"}, {"aliases": [], "benchmark_id": "llm-stats-groundui-1k", "caveat": "A subset of GroundUI-18K for UI grounding evaluation, where models must predict action coordinates on screenshots based on single-step instructions across web, desktop, and mobile platforms.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "GroundUI-1K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 473, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/groundui-1k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-gsm-8k-cot", "caveat": "Grade School Math 8K with Chain-of-Thought prompting, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "GSM-8K (CoT)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 474, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm-8k-%28cot%29?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1272-gsm1k", "caveat": "GSM1k is meant for evaluating LLM's math reasoning ability. It mirrors the style and complexity of the established GSM8k benchmark while consider the problem of data-leaking and overfitting GSM1k可用于评估LLM的数学推理能力；它与GSM8k保持了风格和复杂性的一致，同时考虑了数据泄露和过拟合的问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GSM1k"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "GSM1k", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 475, "released": "2024-05-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GSM1k"}, {"aliases": [], "benchmark_id": "llm-stats-gsm8k", "caveat": "Grade School Math 8K, a dataset of 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "GSM8k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 476, "released": "2021-04-01", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-535-gsm8k", "caveat": "GSM8K is a dataset of 8,500 high quality linguistically diverse grade school math word problems created by human problem writers. The dataset is segmented into 7,500 training problems and 1,000 test problems. These problems take between 2 and 8 steps to solve, and solutions primarily involve performing a sequence of elementary calculations using basic arithmetic operations (+ − × ÷) to reach the final answer. GSM8K 是一个包含 8,500 个高质量、语言多样化的小学数学单词问题的数据集，由人类问题编写者创建。该数据集分为 7,500 个训练问题和 1,000 个测试问题。这些问题的解题步骤在 2 到 8 步之间，解题过程主要涉及使用基本算术运算（+ - × ÷）进行一连串的基本计算，从而得出最终答案。一个聪明的初中生应该能够解决每一个问题。它可用于多步数学推理。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GSM8K"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "GSM8K", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 477, "released": "2021-04-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K"}, {"aliases": [], "benchmark_id": "llm-stats-gsm8k-chat", "caveat": "Grade School Math 8K adapted for chat format evaluation, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "GSM8K Chat", "organization_count": 1, "organizations": ["llm_stats"], "rank": 478, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/gsm8k-chat?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2233-gsm8k-v", "caveat": "GSM8K-V is a multi-image, purely visual mathematical reasoning benchmark built by rendering GSM8K textual problems into visual scenes. It exposes substantial gaps in current vision-language models’ visual reasoning despite their near-saturated performance on text-based GSM8K. GSM8K-V 是一个多图视觉数学推理基准，通过将 GSM8K 的文本题系统性转换为多场景图像构建而成。它揭示了当前 VLMs 在视觉数学推理上的表现与其在文本数学推理上的表现之间存在显著差距。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GSM8K-V"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "GSM8K-V", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 479, "released": "2025-09-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GSM8K-V"}, {"aliases": [], "benchmark_id": "opencompass-1328-gta", "caveat": "GTA is meant for LLM's tool-use evaluations under real-world scenarios, including 229 real-world tasks and executable tool chains. GTA用于评估LLM调用工具解决实际问题的能力，由229个真实任务和可执行工具链组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/GTA"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "GTA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 480, "released": "2024-07-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/GTA"}, {"aliases": [], "benchmark_id": "opencompass-2035-gym4real", "caveat": "Gym4ReaL is a benchmark suite designed to evaluate reinforcement learning algorithms in real-world scenarios, addressing challenges such as non-stationarity, partial observability, and large state-action spaces. Gym4ReaL 是一个用于评估强化学习算法在真实世界场景中表现的基准套件，涵盖非平稳性、部分可观测性和大状态-动作空间等挑战。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Gym4ReaL"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "Gym4ReaL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 481, "released": "2025-06-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Gym4ReaL"}, {"aliases": [], "benchmark_id": "llm-stats-hallusion-bench", "caveat": "A comprehensive benchmark designed to evaluate image-context reasoning in large visual-language models (LVLMs) by challenging models with 346 images and 1,129 carefully crafted questions to assess language hallucination and visual illusion", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Hallusion Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 482, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hallusion-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1355-hallusionbench", "caveat": "HallusionBench is a comprehensive benchmark designed for the evaluation of image-context reasoning, comprising 346 images paired with 1129 questions. HallusionBench是一个专为评估图像上下文推理而设计的综合基准测试，包括346张图像和1129个问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HallusionBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "HallusionBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 483, "released": "2023-10-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HallusionBench"}, {"aliases": [], "benchmark_id": "opencompass-1121-halueval", "caveat": "HaluEval evaluates the performance of LLMs in recognizing hallucination. It includes 5,000 general user queries with ChatGPT responses and 30,000 task-specific examples from three tasks, i.e., question answering, knowledge-grounded dialogue, and text summarization. HaluEval用于评估大语言模型识别幻觉的能力，包含 5,000 条普通用户查询及 ChatGPT 的回答，以及来自三个任务的 30,000 个特定任务示例，即问答、基于知识的对话和文本摘要。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HaluEval"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "HaluEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 484, "released": "2023-10-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HaluEval"}, {"aliases": ["Harbor-Index", "Harbor Index"], "benchmark_id": "harbor_index", "caveat": "Compact high-signal frontier-agent evaluation; the environment and harness define the score.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "coding_agent", "name": "Harbor-Index", "organization_count": 1, "organizations": ["Tencent"], "rank": 485, "released": null, "source": "model_reports", "url": "https://github.com/harbor-framework/harbor-index"}, {"aliases": [], "benchmark_id": "opencompass-1842-hardmath2", "caveat": "HARDMath2 is a benchmark for applied mathematics created by students in a graduate class at Harvard University, featuring 211 original problems covering core topics such as boundary-layer analysis, WKB methods, asymptotic solutions of nonlinear partial differential equations, and the asymptotics. HARDMath2是由哈佛大学研究生课程的学生创建的一项应用数学基准测试，包含211道原创问题，涵盖边界层分析、WKB方法、非线性偏微分方程的渐近解以及振荡积分的渐近性等核心主题。该基准通过一种创新的协作方式构建，学生不仅设计并改进符合课程大纲的高难度问题，还对解决方案进行同行验证，同时测试不同模型的表现。最终，LLM生成的解答会与学生的答案以及数值真值进行自动对比，以评估模型的准确性和能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HARDMath2"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "HARDMath2", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 486, "released": "2025-05-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HARDMath2"}, {"aliases": [], "benchmark_id": "llm-stats-harvey-lab", "caveat": "Harvey LAB (Vals) is a professional legal-work evaluation of AI systems on complex law-firm style tasks, reported by Vals.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500"], "document_share": 0.0008271298593879239, "domain": "knowledge", "name": "Harvey LAB (Vals)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 487, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-harvey-lab-aa", "caveat": "Legal agentic work, criterion pass rate", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/harvey-lab-aa"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "Harvey LAB-AA", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 488, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/harvey-lab-aa"}, {"aliases": [], "benchmark_id": "llm-stats-harvey-lab-aa", "caveat": "Harvey LAB-AA evaluates model performance on complex legal workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "knowledge", "name": "Harvey LAB-AA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 489, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/harvey-lab-aa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-healthbench", "caveat": "An open-source benchmark for measuring performance and safety of large language models in healthcare, consisting of 5,000 multi-turn conversations evaluated by 262 physicians using 48,562 unique rubric criteria across health contexts and behavioral dimensions", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "healthcare", "name": "HealthBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 490, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-healthbench-consensus", "caveat": "HealthBench Consensus is a HealthBench subset focused on questions where physician-created rubric criteria have especially high agreement, measuring healthcare performance and safety on consensus-evaluable conversations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "healthcare", "name": "HealthBench Consensus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 491, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-consensus?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-healthbench-hard", "caveat": "A challenging variation of HealthBench that evaluates large language models' performance and safety in healthcare through 5,000 multi-turn conversations with particularly rigorous evaluation criteria validated by 262 physicians from 60 countries", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "healthcare", "name": "HealthBench Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 492, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-hard?top_n=500"}, {"aliases": ["HealthBench Professional"], "benchmark_id": "healthbench_professional", "caveat": "A distinct, harder split from the HealthBench variants recorded separately in this registry, so its scores are not comparable to them. Published as an OpenAI paper (525 examples, 3 clinician use cases) with a public dataset; released 2026-04-22 per PDF metadata. Graded by an LLM against physician-written rubrics.", "document_count": 1, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5"], "document_share": 0.0008271298593879239, "domain": "health", "name": "HealthBench Professional", "organization_count": 1, "organizations": ["Anthropic"], "rank": 493, "released": "2026-04-22", "source": "model_reports", "url": "https://cdn.openai.com/dd128428-0184-4e25-b155-3a7686c7d744/HealthBench-Professional.pdf"}, {"aliases": [], "benchmark_id": "llm-stats-healthbench-professional", "caveat": "HealthBench Professional evaluates model capability and safety for clinician use cases using real clinician-style chats and physician-authored grading rubrics.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500"], "document_share": 0.0008271298593879239, "domain": "healthcare", "name": "HealthBench Professional", "organization_count": 1, "organizations": ["llm_stats"], "rank": 494, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/healthbench-professional?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-hellaswag", "caveat": "A challenging commonsense natural language inference dataset that uses Adversarial Filtering to create questions trivial for humans (>95% accuracy) but difficult for state-of-the-art models, requiring completion of sentence endings based on physical situations and everyday activities", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HellaSwag", "organization_count": 1, "organizations": ["llm_stats"], "rank": 495, "released": "2019-05-19", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hellaswag?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-531-hellaswag", "caveat": "HellaSwag is a challenge dataset for evaluating commonsense natural language inference, which is specially hard for state-of-the-art models, though its questions are trivial for humans (>95% accuracy). It consists of 70k multiple choice questions, each with a scenario and four possible endings, which requires to select the most reasonable ending. These questions come from two domains:activitynet and wikihow, involving video and text scenarios respectively. The correct answers of these questions are the real sentences for the next event, while the incorrect answers are adversarially generated and human verified, so as to fool machines but not humans. HellaSwag 是一个用于评估常识性自然语言推理的数据集，HellaSwag的问题对于最先进的模型来说是特别困难的，尽管它的问题对于人类来说非常轻松就能回答的（> 95% 的准确率）。它由7万多道多项选择题组成，每道题都有一个场景和四种可能的答案，需要选择最合理的答案。这些问题来自两个领域：activitynet和wikihow，分别涉及视频和文本场景。这些问题的正确答案是下一个事件的真实句子，而错误答案是通过对抗技术生成的并经过人类验证，这些答案可以欺骗机器但不能欺骗人类。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HellaSwag"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "HellaSwag", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 496, "released": "2019-05-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HellaSwag"}, {"aliases": [], "benchmark_id": "opencompass-1097-hellobench", "caveat": "HelloBench is a hierarchical long text generation benchmark to evaluate LLMs' performance in generating long text.  Based on Bloom's Taxonomy, HelloBench categorizes long text generation tasks into five subtasks: open-ended QA, summarization, chat, text completion, and heuristic text generation. HelloBench为长文本生成基准，这是一个全面的、开放式的基准，用于评估LLM在生成长文本方面的性能。基于Bloom的分类法，HelloBench将长文本生成任务分为五个子任务：开放式QA、摘要、聊天、文本完成和启发式文本生成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HelloBench"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "HelloBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 497, "released": "2024-09-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HelloBench"}, {"aliases": [], "benchmark_id": "opencompass-2031-herb", "caveat": "HERB comprises 39,190 artifacts—including documents, meeting transcripts, Slack messages, GitHub content, and URLs—simulating business workflows across product planning, development, and support stages, with noisy, multi-hop QA tasks featuring both answerable and unanswerable queries. HERB基准包含39,190个企业文档、会议记录、Slack消息、GitHub内容和网页链接，模拟产品规划、开发与支持等业务流程，生成包含噪声的多跳问答任务，涵盖可回答与不可回答查询。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HERB"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "HERB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 498, "released": "2025-06-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HERB"}, {"aliases": [], "benchmark_id": "llm-stats-hiddenmath", "caveat": "Google DeepMind's internal mathematical reasoning benchmark that introduces novel problems not encountered during model training to evaluate true mathematical reasoning capabilities rather than memorization", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "HiddenMath", "organization_count": 1, "organizations": ["llm_stats"], "rank": 499, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hiddenmath?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-hipho", "caveat": "HiPhO is a high-school physics olympiad benchmark evaluating multimodal reasoning over physics problems that include diagrams and figures.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "HiPhO", "organization_count": 1, "organizations": ["llm_stats"], "rank": 500, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hipho?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-hle-verified", "caveat": "HLE-Verified evaluates multidisciplinary expert reasoning on a verified subset of Humanity's Last Exam.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HLE-Verified", "organization_count": 1, "organizations": ["llm_stats"], "rank": 501, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hle-verified?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-hmmt-2025", "caveat": "Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "HMMT 2025", "organization_count": 1, "organizations": ["llm_stats"], "rank": 502, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-2025?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-hmmt-feb-26", "caveat": "HMMT February 2026 is a math competition benchmark based on problems from the Harvard-MIT Mathematics Tournament, testing advanced mathematical problem-solving and reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "HMMT Feb 26", "organization_count": 1, "organizations": ["llm_stats"], "rank": 503, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt-feb-26?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-hmmt25", "caveat": "Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "HMMT25", "organization_count": 1, "organizations": ["llm_stats"], "rank": 504, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hmmt25?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1581-holobench", "caveat": "HoloBench is a benchmark designed to evaluate the ability of long-context language models (LCLMs) to perform holistic reasoning over extended text contexts. HoloBench 是一个用于评估长上下文语言模型（LCLMs）在扩展文本上下文中进行整体推理能力的基准测试。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HoloBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "HoloBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 505, "released": "2024-10-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HoloBench"}, {"aliases": [], "benchmark_id": "horizonmath", "caveat": "Unsolved research-level mathematics, reported at pass@4, so figures sit far below conventional accuracy scales.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "math", "name": "HorizonMath", "organization_count": 1, "organizations": ["Tencent"], "rank": 506, "released": null, "source": "model_reports", "url": "https://github.com/ewang26/HorizonMath"}, {"aliases": [], "benchmark_id": "llm-stats-horizonmath", "caveat": "HorizonMath is an extremely difficult frontier mathematics benchmark designed to test the limits of mathematical reasoning on research-level and competition-beyond problems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "HorizonMath", "organization_count": 1, "organizations": ["llm_stats"], "rank": 507, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/horizonmath?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1108-hotpotqa", "caveat": "HOTPOTQA is a dataset with 113k Wikipedia-based question-answer pairs. HotpotQA 用于评估大语言模型的推理能力，包含 113,000 个基于维基百科的问题和答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HotpotQA"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "HotpotQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 508, "released": "2018-09-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HotpotQA"}, {"aliases": [], "benchmark_id": "llm-stats-hr-bench-4k", "caveat": "HR-Bench (4k) evaluates image understanding on high-resolution visual inputs with a 4k setting.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "HR-Bench (4k)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 509, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hr-bench-4k?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1992-htfllib", "caveat": "HtFLlib is a benchmark for heterogeneous federated learning that examines how 40 vision, NLP and sensor models and 10 algorithms collaborate under non-IID data. HtFLlib 是一个面向异构联邦学习算法的综合评测基准，旨在衡量不同模型架构在非 IID 数据环境中的协同学习能力。评测对象覆盖图像、文本与传感信号三类模型，总计 40 个架构及 10 种代表性方法。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HtFLlib"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "HtFLlib", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 510, "released": "2025-06-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HtFLlib"}, {"aliases": [], "benchmark_id": "llm-stats-humaneval", "caveat": "A benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HumanEval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 511, "released": "2021-07-08", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-537-humaneval", "caveat": "The benchmark consists of around 1,000 crowd-sourced Python programming problems, designed to be solvable by entry level programmers, covering programming fundamentals, standard library functionality, and so on. Each problem consists of a task description, code solution and 3 automated test cases. 这是 \"Evaluating Large Language Models Trained on Code\" 论文中描述的 HumanEval 问题解决数据集的评估工具包。它用于测量从文档脚本合成程序的功能正确性。它由 164 个原始编程问题组成，评估语言理解能力、算法和简单数学，其中一些问题与简单的软件面试题类似。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HumanEval"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "HumanEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 512, "released": "2021-07-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval"}, {"aliases": [], "benchmark_id": "llm-stats-humaneval-plus", "caveat": "Enhanced version of HumanEval that extends the original test cases by 80x using EvalPlus framework for rigorous evaluation of LLM-synthesized code functional correctness, detecting previously undetected wrong code", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HumanEval Plus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 513, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-plus?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-humaneval-2", "caveat": "Enhanced version of HumanEval that extends the original test cases by 80x using EvalPlus framework for rigorous evaluation of LLM-synthesized code functional correctness, detecting previously undetected wrong code", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HumanEval+", "organization_count": 1, "organizations": ["llm_stats"], "rank": 514, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval%2B?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-humaneval-average", "caveat": "A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HumanEval-Average", "organization_count": 1, "organizations": ["llm_stats"], "rank": 515, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-average?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-humaneval-er", "caveat": "A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HumanEval-ER", "organization_count": 1, "organizations": ["llm_stats"], "rank": 516, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-er?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-humaneval-mul", "caveat": "A multilingual variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "HumanEval-Mul", "organization_count": 1, "organizations": ["llm_stats"], "rank": 517, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humaneval-mul?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-543-humaneval-x", "caveat": "HumanEval-X is a benchmark for evaluating the multilingual ability of code generative models. It consists of 820 high-quality human-crafted data samples (each with test cases) in Python, C++, Java, JavaScript, and Go, and can be used for various tasks, such as code generation and translation. HumanEval-X 是一个用于评估代码生成模型的多语言能力的基准测试。它包含了820个高质量的人工制作的数据样本（每个都有测试案例），包括Python、C++、Java、JavaScript和Go语言，可用于各种任务，如代码生成和翻译。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HumanEval-X"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "HumanEval-X", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 518, "released": "2023-03-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HumanEval-X"}, {"aliases": [], "benchmark_id": "llm-stats-humanevalfim-average", "caveat": "Average evaluation of HumanEval Fill-in-the-Middle benchmark variants (single-line, multi-line, random-span) for assessing code infilling capabilities of language models", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "HumanEvalFIM-Average", "organization_count": 1, "organizations": ["llm_stats"], "rank": 519, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanevalfim-average?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-humanity-s-last-exam", "caveat": "Humanity's Last Exam (HLE) is a multi-modal academic benchmark with 2,500 questions across mathematics, humanities, and natural sciences, designed to test LLM capabilities at the frontier of human knowledge with unambiguous, verifiable solutions", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "Humanity's Last Exam", "organization_count": 1, "organizations": ["llm_stats"], "rank": 520, "released": "2025-01-24", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-humanity-s-last-exam-no-tools-text-only", "caveat": "Text-only Humanity's Last Exam variant evaluated without tool use.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "Humanity's Last Exam (no tools, text-only)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 521, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28no-tools%2C-text-only%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-humanity-s-last-exam-with-tools-text-only", "caveat": "Text-only Humanity's Last Exam variant evaluated with tool use enabled.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "Humanity's Last Exam (with tools, text-only)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 522, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/humanity%27s-last-exam-%28with-tools%2C-text-only%29?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-humanitys-last-exam", "caveat": "Reasoning & knowledge", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/humanitys-last-exam"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "Humanity’s Last Exam", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 523, "released": "2025-01-23", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam"}, {"aliases": [], "benchmark_id": "llm-stats-hypersim", "caveat": "Hypersim evaluates 3D grounding and depth understanding in synthetic indoor scenes.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "Hypersim", "organization_count": 1, "organizations": ["llm_stats"], "rank": 524, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/hypersim?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1782-hypobench", "caveat": "HypoBench, a novel benchmark designed to evaluate LLMs and hypothesis generation methods across multiple aspects, including practical utility, generalizability, and hypothesis discovery rate. HypoBench includes 7 real-world tasks and 5 synthetic tasks with 194 distinct datasets. HypoBench，这是一种新颖的基准，旨在从多个方面评估 LLM 和假设生成方法，包括实用性、泛化性和假设发现率。HypoBench 包括 7 个真实任务和 5 个合成任务，具有 194 个不同的数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HypoBench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "HypoBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 525, "released": "2025-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HypoBench"}, {"aliases": [], "benchmark_id": "opencompass-1786-hypoeval", "caveat": "HypoEval, Hypothesis-guided Evaluation framework, which first uses a small corpus of human evaluations to generate more detailed rubrics for human judgments and then incorporates a checklist-like approach to combine LLM's assigned scores on each decomposed dimension to acquire overall scores. HypoEval，即假设指导的评估框架，该框架首先使用一小部分人工评估来生成更详细的人类判断量规，然后采用类似清单的方法，将 LLM 在每个分解维度上的分配分数结合起来，以获得总分。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/HypoEval"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "HypoEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 526, "released": "2025-04-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/HypoEval"}, {"aliases": [], "benchmark_id": "opencompass-1320-iac-eval", "caveat": "IaC-Eval is meant for quantitatively evaluating the capabilities of LLMs in cloud IaC code generation, containing 458 questions ranging from simple to difficult across various cloud services. IaC-Eval用于定量评估LLM在云IaC代码生成中的功能，其中包含458个从易到难的问题，涵盖了各种云服务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IaC-Eval"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "IaC-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 527, "released": "2024-09-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IaC-Eval"}, {"aliases": [], "benchmark_id": "llm-stats-if", "caveat": "Instruction-Following Evaluation (IFEval) benchmark for large language models, focusing on verifiable instructions with 25 types of instructions and around 500 prompts containing one or more verifiable constraints", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500"], "document_share": 0.0008271298593879239, "domain": "structured_output", "name": "IF", "organization_count": 1, "organizations": ["llm_stats"], "rank": 528, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/if?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-ifbench", "caveat": "Instruction following", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/ifbench"], "document_share": 0.0008271298593879239, "domain": "instruction-following", "name": "IFBench", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 529, "released": "2025-07-03", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/ifbench"}, {"aliases": [], "benchmark_id": "llm-stats-ifbench", "caveat": "Instruction Following Benchmark evaluating model's ability to follow complex instructions", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "instruction_following", "name": "IFBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 530, "released": "2025-07-03", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ifbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ifeval", "caveat": "Instruction-Following Evaluation (IFEval) benchmark for large language models, focusing on verifiable instructions with 25 types of instructions and around 500 prompts containing one or more verifiable constraints", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "structured_output", "name": "IFEval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 531, "released": "2023-11-14", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ifeval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1136-ifeval", "caveat": "IFEval is a straightforward and easy-to reproduce evaluation benchmark. It focuses on a set of “verifiable instructions” such as “write in more than 400 words” and “mention the keyword of AI at least 3 times”. IFEval 是一个简单且易于复现的评估基准。它关注一组“可验证的指令”，例如“写超过 400 个单词”和“至少提到关键词 AI 3 次”。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IFEval"], "document_share": 0.0008271298593879239, "domain": "指令跟随", "name": "IFEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 532, "released": "2023-11-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IFEval"}, {"aliases": [], "benchmark_id": "opencompass-1617-ifir", "caveat": "IFIR is the first comprehensive benchmark designed to evaluate instruction-following information retrieval (IR) in expert domains. IFIR includes 2,426 high-quality examples and covers eight subsets across four specialized domains: finance, law, healthcare, and science literature. IFIR，这是第一个旨在评估专家领域指令跟随信息检索（IR）的综合基准。IFIR 包含 2,426 个高质量示例，涵盖四个专业领域（金融、法律、医疗保健和科学文献）的八个子集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IFIR"], "document_share": 0.0008271298593879239, "domain": "指令跟随", "name": "IFIR", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 533, "released": "2025-03-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IFIR"}, {"aliases": [], "benchmark_id": "llm-stats-image2floorplan", "caveat": "Image2FloorPlan is an in-house benchmark evaluating multimodal models on generating structured floor plans and interactive frontends directly from images such as design mockups and room photos.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Image2FloorPlan", "organization_count": 1, "organizations": ["llm_stats"], "rank": 534, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/image2floorplan?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-imagemining", "caveat": "ImageMining evaluates multimodal models on extracting structured information from images using tool use, measuring ability to combine visual understanding with tool-based retrieval and analysis.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ImageMining", "organization_count": 1, "organizations": ["llm_stats"], "rank": 535, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/imagemining?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1274-imdl-benco", "caveat": "IMDL-BenCo offers a comprehensive IMDL benchmark and modular codebase. IMDL-BenCo提供了全面的IMDL基准测试和模块化代码库。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IMDL-BenCo"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "IMDL-BenCo", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 536, "released": "2024-06-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IMDL-BenCo"}, {"aliases": [], "benchmark_id": "llm-stats-imo-2025", "caveat": "IMO 2025 evaluates models on the six problems from the 2025 International Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "IMO 2025", "organization_count": 1, "organizations": ["llm_stats"], "rank": 537, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-2025?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-imo-answerbench", "caveat": "IMO-AnswerBench is a benchmark for evaluating mathematical reasoning capabilities on International Mathematical Olympiad (IMO) problems, focusing on answer generation and verification.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "IMO-AnswerBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 538, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/imo-answerbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-imoproof-adv", "caveat": "IMOProof-Adv is an advanced benchmark of International Mathematical Olympiad-style proof problems requiring rigorous multi-step mathematical proofs.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "IMOProof-Adv", "organization_count": 1, "organizations": ["llm_stats"], "rank": 539, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/imoproof-adv?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-include", "caveat": "Include benchmark - specific documentation not found in official sources", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "Include", "organization_count": 1, "organizations": ["llm_stats"], "rank": 540, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/include?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-community-07c9946d-dcf0-4977-a640-a6b1356b4f0b", "caveat": null, "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A07c9946d-dcf0-4977-a640-a6b1356b4f0b?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "independence-bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 541, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A07c9946d-dcf0-4977-a640-a6b1356b4f0b?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1670-indicmmlu-pro", "caveat": "IndicMMLU-Pro provides a standardized evaluation framework to push the research boundaries in Indic language AI, facilitating the development of more accurate, efficient, and culturally sensitive models. IndicMMLU-Pro 提供了一个标准化的评估框架，以推动印度语系语言 AI 的研究边界，促进更准确、高效和具有文化敏感性的模型的发展。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IndicMMLU-Pro"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "IndicMMLU-Pro", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 542, "released": "2025-01-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IndicMMLU-Pro"}, {"aliases": [], "benchmark_id": "opencompass-1322-infibench", "caveat": "InfiBench is a large-scale freeform question-answering (QA) benchmark for code to our knowledge, comprising 234 carefully selected high-quality Stack Overflow questions that span across 15 programming languages. InfiBench用于评测LLM回答代码相关问题的能力，包括涵盖15种编程语言的234个精心挑选的高质量Stack Overflow问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InfiBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "InfiBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 543, "released": "2024-03-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/InfiBench"}, {"aliases": [], "benchmark_id": "llm-stats-infinitebench-en-mc", "caveat": "InfiniteBench English Multiple Choice variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "InfiniteBench/En.MC", "organization_count": 1, "organizations": ["llm_stats"], "rank": 544, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.mc?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-infinitebench-en-qa", "caveat": "InfiniteBench English Question Answering variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "InfiniteBench/En.QA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 545, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/infinitebench-en.qa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1085-infobench", "caveat": "InfoBench is a benchmark comprising 500 diverse instructions and 2,250 decomposed questions across multiple constraint categories. InfoBench 是一个指令追随评测基准，包含 500 条多样化的指令和 2,250 个分解问题，涵盖多个约束类别。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InfoBench"], "document_share": 0.0008271298593879239, "domain": "指令跟随", "name": "InfoBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 546, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/InfoBench"}, {"aliases": [], "benchmark_id": "llm-stats-infographicsqa", "caveat": "InfographicVQA dataset with 5,485 infographic images and over 30,000 questions requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "InfographicsQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 547, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/infographicsqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-infovqa", "caveat": "InfoVQA dataset with 30,000 questions and 5,000 infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "InfoVQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 548, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-infovqatest", "caveat": "InfoVQA test set with infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "InfoVQAtest", "organization_count": 1, "organizations": ["llm_stats"], "rank": 549, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/infovqatest?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-instruct-humaneval", "caveat": "Instruction-based variant of HumanEval benchmark for evaluating large language models' code generation capabilities with functional correctness using pass@k metric on programming problems", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "Instruct HumanEval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 550, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/instruct-humaneval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1149-instrusum", "caveat": "InstruSum evaluates the task of instruction controllable text summarization, where the model input consists of both a source article and a natural language requirement for desired summary characteristics. InstruSum 是评测指令可控的文本摘要任务，模型输入包括源文章和对所需摘要特征的自然语言要求。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InstruSum"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "InstruSum", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 551, "released": "2024-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/InstruSum"}, {"aliases": [], "benchmark_id": "llm-stats-intergps", "caveat": "Interpretable Geometry Problem Solver (Inter-GPS) with Geometry3K dataset of 3,002 geometry problems with dense annotation in formal language using theorem knowledge and symbolic reasoning", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "InterGPS", "organization_count": 1, "organizations": ["llm_stats"], "rank": 552, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/intergps?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-internal-api-instruction-following-hard", "caveat": "Internal API instruction following (hard) benchmark - specific documentation not found in official sources", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "structured_output", "name": "Internal API instruction following (hard)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 553, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-api-instruction-following-%28hard%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-internal-research-debugging-evaluation", "caveat": "The Internal Research Debugging Evaluation measures whether models can debug 41 real bugs from internal OpenAI research experiments (plus alignment-auditing tasks), where the original solutions took experienced researchers hours to days. Passing corresponds to providing assistance that would unblock the user, including partial root-cause explanations or fixes.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Internal Research Debugging Evaluation", "organization_count": 1, "organizations": ["llm_stats"], "rank": 554, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/internal-research-debugging-evaluation?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2151-interndata-a1", "caveat": "InternData-A1: A hybrid synthetic-real manipulation dataset integrating 5 heterogeneous robots, 15 skills, and 200+ scenes, emphasizing multi-robot collaboration under dynamic scenarios. InternData-A1：一个融合了 5 种异构机器人、15 项技能和 200+ 场景的混合合成-真实操作数据集，重点关注动态场景下的多机器人协作。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InternData-A1"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "InternData-A1", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 555, "released": "2025-07-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/InternData-A1"}, {"aliases": [], "benchmark_id": "opencompass-2153-interndata-m1", "caveat": "InternData-M1: A large-scale synthetic dataset for generalizable pick-and-place over 80K objects, with open-ended instructions covering object recognition, spatial and commonsense reasoning, and long-horizon tasks. InternData-M1：一个大规模合成数据集，用于可泛化的抓取与放置任务，涵盖 8 万+ 对象，配有开放式指令，涉及物体识别、空间与常识推理以及长时任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InternData-M1"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "InternData-M1", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 556, "released": "2025-07-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/InternData-M1"}, {"aliases": [], "benchmark_id": "opencompass-2152-interndata-n1", "caveat": "InternData-N1: A high-quality navigation dataset with the most diverse scenes and extensive randomization across embodiments/viewpoints, including 3k+ scenes and 830k VLN data. InternData-N1：一个高质量导航数据集，具有最丰富的场景和跨具身/视角的广泛随机化，包含 3000+ 场景和 83 万条 VLN 数据。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/InternData-N1"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "InternData-N1", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 557, "released": "2025-07-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/InternData-N1"}, {"aliases": [], "benchmark_id": "opencompass-1976-intphys2", "caveat": "IntPhys2 tests intuitive physics—permanence, immutability, continuity and solidity—using synthetic videos. SOTA models perform around chance (~50 %), far below human level. IntPhys 2，一个用于评估深度学习模型直观物理理解能力的视频基准。它围绕永恒性、不可变性、时空连续性和实体性四个核心原则，测试模型区分可能与不可能事件的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IntPhys2"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "IntPhys2", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 558, "released": "2025-06-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IntPhys2"}, {"aliases": [], "benchmark_id": "llm-stats-ipho-2025", "caveat": "International Physics Olympiad 2025 (theory) comprises all 3 theory problems from the official 2025 IPhO competition. Results are based on blinded human evaluation with guidelines based on the official competition scoring, validated by domain experts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500"], "document_share": 0.0008271298593879239, "domain": "physics", "name": "IPhO 2025", "organization_count": 1, "organizations": ["llm_stats"], "rank": 559, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ipho-2025?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1840-iqbench", "caveat": "IQBench is a novel benchmark designed to evaluate the fluid intelligence of VisionLanguage Models (VLMs) using standardized visual IQ tests. It consists of 500 manually collected and annotated visual IQ questions covering various domains. IQBench是一个新基准测试，旨在通过标准化视觉智商测试评估视觉语言模型（VLMs）的流体智力。该基准包含500个手动收集和注释的视觉智商问题，涵盖模式识别、类比推理、视觉算术、空间理解等多个领域。与以往仅关注最终答案准确性的基准不同，IQBench强调对模型推理能力的评估，采用双重评估框架：准确性评分和推理评分。实验表明，即使是性能最高的模型（如o4mini、gemini2.5flash和claude3.7sonnet），在3D空间和字母重组任务上也表现出明显不足，凸显了当前VLMs在通用推理能力上的局限性。IQBench为开发更透明、更具认知能力的多模态系统奠定了基础。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IQBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "IQBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 560, "released": "2025-05-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IQBench"}, {"aliases": [], "benchmark_id": "opencompass-2091-is-bench", "caveat": "IS-Bench is the first evaluation benchmark dedicated to assessing the safety of embodied agents during their interaction with home environments. It encompasses 161 high-risk domestic scenarios, spanning 10 hazard categories, including food poisoning, fire, electric shock, and more. IS-Bench 是首个专注于具身智能体与家用环境交互过程安全性的评测基准。它包含 161 个高风险家居场景及相关日常任务，覆盖食物中毒、火灾、触电等 10 大类常见风险。通过贯穿整个交互过程的动态评测框架，全方位评估具身智能体的安全素养。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/IS-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "IS-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 561, "released": "2025-07-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/IS-Bench"}, {"aliases": [], "benchmark_id": "opencompass-2077-itbench", "caveat": "ITBench is a benchmark designed to evaluate the performance of AI agents in real-world IT automation tasks. ITBench 是一个旨在评估 AI 智能体在真实世界 IT 自动化任务中表现的基准。它涵盖了站点可靠性工程、合规与安全运营以及财务运营等关键维度，并包含 102 个真实场景。该基准提供了一个开源框架和多种基线智能体实现，并集成了CrewAI等工具，以促进AI驱动的IT自动化发展。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ITBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "ITBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 562, "released": "2025-02-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ITBench"}, {"aliases": [], "benchmark_id": "artificial-analysis-itbench-aa", "caveat": "Kubernetes incident root-cause analysis", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/itbench-aa"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "ITBench-AA", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 563, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/itbench-aa"}, {"aliases": [], "benchmark_id": "opencompass-2073-j1-bench", "caveat": "J1-Bench is an interactive and comprehensive legal benchmark where LLM agents engage in diverse legal scenarios, completing tasks through interactions with various participants under procedural rules. J1-Bench 是一个交互式的综合法律基准，法律智能体在此参与各种法律情景，根据程序规则通过与不同参与者的互动完成指定的法律任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/J1-Bench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "J1-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 564, "released": "2025-07-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/J1-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1337-jailtrickbench", "caveat": "JailTrickBench can evaluate the impact of various attack settings on LLM performance, including 8 key factors of implementing jailbreak attacks on LLMs from both target-level and attack-level perspectives. JailTrickBench用于评估LLM应对各种越狱攻击的能力，涵盖从目标级和攻击级2个角度实施越狱攻击的8个关键因素。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JailTrickBench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "JailTrickBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 565, "released": "2024-06-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/JailTrickBench"}, {"aliases": [], "benchmark_id": "opencompass-1554-jl1-cd", "caveat": "JL1-CD is a large-scale, sub-meter, all-inclusive open-source dataset for remote sensing image change detection (CD). It contains 5,000 pairs of 512×512 pixel satellite images with a resolution of 0.5 to 0.75 meters, covering various types of surface changes in multiple regions of China. JL1-CD 是一个大规模、亚米级、全包含的开源遥感影像变化检测（CD）数据集。它包含 5000 对 512×512 像素的卫星影像，分辨率为 0.5 至 0.75 米，覆盖中国多个地区的各种地表变化。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JL1-CD"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "JL1-CD", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 566, "released": "2025-02-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/JL1-CD"}, {"aliases": [], "benchmark_id": "llm-stats-job-bench", "caveat": "Job Bench evaluates AI agents on realistic professional tasks that require multi-step planning, research, and production of work artifacts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "productivity", "name": "Job Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 567, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/job-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2130-job-complex", "caveat": "JOB-Complex is a challenging benchmark for traditional and learned query optimizers, containing 30 SQL queries and a plan-selection benchmark with nearly 6000 execution plans. JOB-Complex是一个面向传统与学习式查询优化器的挑战性数据库基准，包含30个SQL查询和近6000个执行计划的计划选择基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JOB-Complex"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "JOB-Complex", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 568, "released": "2025-07-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/JOB-Complex"}, {"aliases": [], "benchmark_id": "jobbench", "caveat": "Professional job tasks aligned with human work; rubric grading moves the number beyond model capability alone.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "professional", "name": "JobBench", "organization_count": 1, "organizations": ["Tencent"], "rank": 569, "released": "2026-05-25", "source": "model_reports", "url": "https://arxiv.org/abs/2605.26329"}, {"aliases": [], "benchmark_id": "opencompass-1583-judgebench", "caveat": "JudgeBench is a benchmark aimed at evaluating LLM-based judges for objective correctness on challenging response pairs. JudgeBench 是一个旨在评估基于LLM的裁判在具有挑战性的响应对上的客观正确性的基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/JudgeBench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "JudgeBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 570, "released": "2024-10-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/JudgeBench"}, {"aliases": [], "benchmark_id": "opencompass-2076-k-sort-arena", "caveat": "K-Sort Arena employs K-wise comparison, allowing K models to participate in a free-for-all, providing richer information than pairwise comparison. It also designs a matching algorithm based on exploration-exploitation and probabilistic modeling to achieve more efficient and reliable model ranking. 本项目提出K-Sort Arena，采用 K-wise 比较，允许 K 个模型参与自由混战，提供比成对比较更丰富的信息，并设计基于探索-利用的匹配算法和概率建模，从而实现更高效和更可靠的模型排名。目前，K-Sort Arena 已收集几千次高质量投票并构建了全面的模型排行榜，已用于评估几十种最先进的视觉生成模型，包括文生图和文生视频模型。K-Sort Arena已经历数月的项目内测，期间收到来自加州大学伯克利分校, 新加坡国立大学, 卡内基梅隆大学, 斯坦福大学, 普林斯顿大学, 北京大学等数十家机构的专业人员的技术反馈，现已公开线上发布。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/K-Sort-Arena"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "K-Sort-Arena", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 571, "released": "2024-09-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/K-Sort-Arena"}, {"aliases": [], "benchmark_id": "llm-stats-kernel-bench-l3", "caveat": "Kernel Bench L3 evaluates agentic GPU kernel optimization across 50 problems. Qwen reports two metrics for this benchmark: median per-problem speedup over the PyTorch eager reference and the fraction of problems faster than torch.compile.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Kernel Bench L3", "organization_count": 1, "organizations": ["llm_stats"], "rank": 572, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/kernel-bench-l3?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-kernelbench-hard", "caveat": "KernelBench Hard evaluates agentic GPU kernel optimization on the hardest problem set. Each question is scored by the agent's submitted operator TFLOPs relative to the theoretical peak of the current hardware, with the benchmark score being the average across all questions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "KernelBench Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 573, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelbench-hard?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-kernelgen-1p", "caveat": "KernelGen 1P is an OpenAI AI-self-improvement evaluation that measures whether models can write and optimize compute kernels, part of the suite tracking progress toward accelerating internal research.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "KernelGen 1P", "organization_count": 1, "organizations": ["llm_stats"], "rank": 574, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/kernelgen-1p?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-kimi-claw-24-7-bench", "caveat": "Kimi Claw 24/7 Bench is Moonshot AI's in-house benchmark for evaluating long-horizon agentic performance in persistent, multi-day coworking tasks. It spans 17 professional scenarios across 610 evaluation points, covering software engineering, ML research, recruiting, trading, and marketing tasks executed through the OpenClaw harness.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Kimi Claw 24/7 Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 575, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-claw-24-7-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-kimi-code-bench-v2", "caveat": "Kimi Code Bench v2 is Moonshot AI's in-house benchmark for evaluating coding agents on realistic software engineering tasks across 10+ mainstream programming languages and a production tech stack spanning backend services, infrastructure, performance engineering, systems programming, security, frontend development, and ML/data engineering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Kimi Code Bench v2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 576, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/kimi-code-bench-v2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-kina", "caveat": "KINA is a knowledge-intensive evaluation that measures a model's breadth and depth of factual knowledge across academic and professional domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "KINA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 577, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/kina?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1543-kitab-bench", "caveat": "KITAB-Bench is a Comprehensive Multi-Domain Benchmark for Arabic OCR and Document Understanding,  and spans 36 sub-domains with over 8,809 samples, carefully curated to rigorously evaluate essential skills required for Arabic OCR and document analysis. KITAB-Bench是一个全面多领域阿拉伯文 OCR 和文档理解基准，包含 36 个子领域，超过 8,809 个样本，经过精心挑选，以严格评估阿拉伯 OCR 和文档分析所需的基本技能。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KITAB-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "KITAB-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 578, "released": "2025-02-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/KITAB-Bench"}, {"aliases": [], "benchmark_id": "opencompass-2126-kmmlu-redux", "caveat": "KMMLU-Redux is a reconstructed version of the existing KMMLU, comprising 2,587 problems from Korean National Technical Qualification (KNTQ) exams. KMMLU-Redux是现有 KMMLU 的一个重建版本，包含来自韩国国家技术资格（KNTQ）考试的 2,587 个问题。我们发现了 KMMLU 中的一些关键问题，包括泄露的答案、缺乏清晰度、问题表述不当、符号错误和污染风险。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KMMLU-Redux"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "KMMLU-Redux", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 579, "released": "2025-07-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/KMMLU-Redux"}, {"aliases": [], "benchmark_id": "opencompass-1635-knowlogic", "caveat": "KnowLogic is a knowledge-driven synthetic benchmark designed to evaluate the reasoning abilities of large language models (LLMs). It includes 5400 bilingual (Chinese and English) questions across various domains, covering different aspects of commonsense knowledge and logical reasoning. KnowLogic 是一个以知识驱动的合成基准，旨在评估大型语言模型的推理能力（LLMs）。它包含涵盖各个领域、涵盖常识知识和逻辑推理不同方面的 5400 个中英双语问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KnowLogic"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "KnowLogic", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 580, "released": "2025-03-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/KnowLogic"}, {"aliases": [], "benchmark_id": "opencompass-1706-koffvqa", "caveat": "KOFFVQA is a carefully crafted free-form visual question answering(VQA) benchmark in the Korean language consisting of 275 questions across 10 different tasks. KOFFVQA是一个精心设计的韩语自由形式视觉问答（VQA）基准测试，包含10个不同任务中的275个问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KOFFVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "KOFFVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 581, "released": "2025-03-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/KOFFVQA"}, {"aliases": [], "benchmark_id": "opencompass-1245-kor-bench", "caveat": "Knowledge-Orthogonal Reasoning Benchmark (KOR-Bench) encompasses five task categories: Operation, Logic, Cipher, Puzzle, and Counterfactual. KOR-Bench emphasizes the effectiveness of models in applying new rule descriptions to solve novel rule-driven questions. KOR-Bench用于评估大语言模型的推理能力，包括五个任务类别：操作、逻辑、密码、拼图和反事实。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/KOR-Bench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "KOR-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 582, "released": "2024-10-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/KOR-Bench"}, {"aliases": [], "benchmark_id": "opencompass-541-l-eval", "caveat": "L-Eval is a comprehensive Long Context Language Models (LCLMs) evaluation suite with 20 sub-tasks, 508 long documents, and over 2,000 human-labeled query-response pairs encompassing diverse question styles, domains, and input length (3k～200k tokens). L-Eval 是一个全面的长上下文语言模型（LCLMs）评估套件，包括 20 个子任务、508 个长文档和超过 2,000 个人工标记的查询-响应对。它涵盖了多种问答风格、领域和输入长度（3,000 至 200,000 个 token）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/L-Eval"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "L-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 583, "released": "2023-10-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/L-Eval"}, {"aliases": [], "benchmark_id": "llm-stats-labbench2", "caveat": "LABBench2 evaluates models on real-world biology research tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LABBench2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 584, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/labbench2?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-523-lambada", "caveat": "The LAMBADA evaluates the capabilities of computational models for text understanding by means of a word prediction task. LAMBADA is a collection of narrative passages sharing the characteristic that human subjects are able to guess their last word if they are exposed to the whole passage, but not if they only see the last sentence preceding the target word. To succeed on LAMBADA, computational models cannot simply rely on local context, but must be able to keep track of information in the broader discourse.\nThe LAMBADA dataset is extracted from BookCorpus and consists of 10'022 passages, divided into 4'869 development and 5'153 test passages, comprising 203 million words. LAMBADA 通过一个单词预测任务来评估计算模型对文本理解的能力。LAMBADA 是有如下特点的一组叙述性文章：如果面对整篇文章，人们可以猜测它们的最后一个单词，但如果他们只看到目标单词前面的最后一句话，就无法猜测。为了在 LAMBADA 上由好的效果，模型不能仅仅依赖于局部上下文，而必须能够跟踪更广泛的话语信息。\nLAMBADA 数据集是从 BookCorpus 中提取的，包括 10,022 段落，分为 4,869 个开发段落和 5,153 个测试段落，共计 2.03 亿个单词。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LAMBADA"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "LAMBADA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 585, "released": "2016-06-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LAMBADA"}, {"aliases": [], "benchmark_id": "opencompass-1924-lamp-qa", "caveat": "The benchmark covers questions from three major categories: (1) Arts & Entertainment, (2) Lifestyle & Personal Development, and (3) Society & Culture, encompassing over 45 subcategories in total. 基准旨在评估个性化长篇答案生成。该基准涵盖三大类问题:(1)艺术与娱乐,(2)生活与个人发展,(3)社会与文化,共包含45个以上的子类别。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LaMP-QA"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "LaMP-QA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 586, "released": "2025-05-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LaMP-QA"}, {"aliases": [], "benchmark_id": "opencompass-2129-langnavbench", "caveat": "LangNavBench is a benchmark for evaluating natural language understanding in semantic navigation, built upon the manually verified LangNav open-set dataset. LangNavBench是一个语义导航中自然语言理解评测基准，基于手工验证的LangNav开放集数据集，评测具身智能体在自然语言指令引导下的目标定位能力，涵盖类别层次理解、对象属性识别和空间关系推理等维度.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LangNavBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "LangNavBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 587, "released": "2025-07-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LangNavBench"}, {"aliases": [], "benchmark_id": "opencompass-2089-lara", "caveat": "LaRA is a focused benchmark for testing Retrieval-Augmented Generation (RAG) and long-context LLMs. LaRA 是专为评估检索增强生成（RAG）和长上下文大型语言模型（LLM）而打造的基准。它围绕信息定位、片段对比、内容推理和幻觉检测四大能力，通过 2 326 条测试用例，对四类问答任务和三种长文本场景（小说、学术论文、财务报表）进行全面测评。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LaRA"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "LaRA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 588, "released": "2025-02-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LaRA"}, {"aliases": [], "benchmark_id": "llm-stats-lbpp-v2", "caveat": "LBPP (v2) benchmark - specific documentation not found in official sources, possibly related to language-based planning problems", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LBPP (v2)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 589, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/lbpp-%28v2%29?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-520-lcsts", "caveat": "LCSTS is a large corpus of Chinese short text summarization dataset constructed from the Chinese microblogging website Sina Weibo, which is released to the public. This corpus consists of over 2 million real Chinese short texts with short summaries given by the author of each text. LCSTS是一个大规模的中文短文本摘要数据集，从中国微博网站新浪微博中构建而成，并已开源。该数据集包含超过 200 万条真实的中文短文本，每个文本都提供了一个简短的摘要。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LCSTS"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "LCSTS", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 590, "released": "2015-06-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LCSTS"}, {"aliases": ["Legal Agent Benchmark"], "benchmark_id": "legal_agent_benchmark", "caveat": "Open-source agent benchmark from Harvey (github.com/harveyai/harvey- labs): 1,200+ tasks across 24 practice areas, all-pass LLM-judge grading. Absolute scores are very low across all models (0-13%), so ordering at the top is decided by a handful of tasks and small gaps are noise.", "document_count": 1, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "Legal Agent Benchmark", "organization_count": 1, "organizations": ["Anthropic"], "rank": 591, "released": "2026-05-06", "source": "model_reports", "url": "https://www.harvey.ai/blog/introducing-harveys-legal-agent-benchmark"}, {"aliases": [], "benchmark_id": "llm-stats-legal-agent-benchmark", "caveat": "The Legal Agent Benchmark (LAB) is Harvey's open-source benchmark for evaluating AI agents on complex, long-horizon legal work. Tasks are scored under an all-pass standard against expert-curated rubrics, where a task passes only if every required rubric criterion (facts, conclusions, citations, structure, and analytical moves) passes.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "Legal Agent Benchmark", "organization_count": 1, "organizations": ["llm_stats"], "rank": 592, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/legal-agent-benchmark?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2422-lens", "caveat": "A multi-level evaluation benchmark of multimodal reasoning with with 3.4K contemporary images and 60K+ human-authored questions covering eight tasks and 12 daily scenarios, forming three progressive task tiers, i.e., perception, understanding, and reasoning. 多层次多模态推理评测基准", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LENS"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "LENS", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 593, "released": "2025-06-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LENS"}, {"aliases": [], "benchmark_id": "llm-stats-lifescibench", "caveat": "LifeSciBench is an expert-authored, expert-reviewed benchmark of 750 open-ended life-science research tasks spanning seven workflows and seven biological domains. Responses are graded against 19,020 physician- and scientist-written rubric criteria rather than multiple-choice answers, and most tasks require interpreting attached artifacts such as figures, PDFs, and sequence files.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "LifeSciBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 594, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/lifescibench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1238-lingoly", "caveat": "Aiming at evaluating LLM's advanced reasoning abilities, LingOly is composed of  olympiad-level linguistic reasoning puzzles in low-resource and extinct languages. It covers more than 90 mostly low-resource languages, and contains 1,133 problems across 6 formats and 5 levels of human difficulty. LingOly由低资源和已灭绝语言的奥林匹克级别语言推理谜题组成，用于评估大语言模型的高级推理能力。其中涵盖了90多种语言，共有1133个涉及6种格式和5个人工难度级别的问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LINGOLY"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "LINGOLY", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 595, "released": "2024-06-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LINGOLY"}, {"aliases": [], "benchmark_id": "llm-stats-lingoqa", "caveat": "A benchmark for multimodal spatial-language understanding and visual-linguistic question answering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "LingoQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 596, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/lingoqa?top_n=500"}, {"aliases": ["LiveBench"], "benchmark_id": "livebench", "caveat": "Contents change by release date; each date is a different benchmark.", "document_count": 1, "document_ids": ["model_reports:qwen3_technical_report"], "document_share": 0.0008271298593879239, "domain": "general", "name": "LiveBench", "organization_count": 1, "organizations": ["Qwen"], "rank": 597, "released": "2024-06-06", "source": "model_reports", "url": "https://livebench.ai/"}, {"aliases": [], "benchmark_id": "llm-stats-livebench", "caveat": "LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "LiveBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 598, "released": "2024-06-12", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1246-livebench", "caveat": "LiveBench contains questions that are based on recently-released math competitions, arXiv papers, news articles, and datasets, and it contains harder, contamination-free versions of tasks from previous benchmarks such as Big-Bench Hard, AMPS, and IFEval. LiveBench是一个LLM基准测试，涵盖数学、编码、推理、语言、指令遵循和数据分析，包含基于最近发布的数学竞赛、arXiv 论文、新闻文章和数据集的问题，及经典基准测试（如 Big-Bench Hard、AMPS 和 IFEval）的更难、无污染的任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveBench"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "LiveBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 599, "released": "2024-06-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveBench"}, {"aliases": [], "benchmark_id": "llm-stats-livebench-20241125", "caveat": "LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "LiveBench 20241125", "organization_count": 1, "organizations": ["llm_stats"], "rank": 600, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livebench-20241125?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-livecodebench", "caveat": "Coding", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/livecodebench"], "document_share": 0.0008271298593879239, "domain": "coding", "name": "LiveCodeBench", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 601, "released": "2024-03-12", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/livecodebench"}, {"aliases": [], "benchmark_id": "llm-stats-livecodebench", "caveat": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LiveCodeBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 602, "released": "2024-03-12", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1413-livecodebench", "caveat": "LiveCodeBench evaluates LLMs' coding abilities. It continuously collects new problems over time from contests across LeetCode, AtCoder, and CodeForces. Notably, it also focuses on a broader range of code related capabilities besides code generation. LiveCodeBench用于评估大语言模型的代码能力，包含来自LeetCode、AtCoder和CodeForces的动态更新的问题，并在代码生成能力的基础上将更广泛的相关能力纳入考量。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveCodeBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "LiveCodeBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 603, "released": "2024-03-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveCodeBench"}, {"aliases": ["LiveCodeBench Pro", "LCB Pro"], "benchmark_id": "livecodebench_pro", "caveat": "Reported as an Elo rating against competitive-programming problems, not a pass rate, so it cannot be read on the same axis as LiveCodeBench.", "document_count": 1, "document_ids": ["model_reports:google_gemini_3_1_pro_model_card"], "document_share": 0.0008271298593879239, "domain": "coding", "name": "LiveCodeBench Pro", "organization_count": 1, "organizations": ["Google"], "rank": 604, "released": "2025-06-13", "source": "model_reports", "url": "https://livecodebenchpro.com/"}, {"aliases": [], "benchmark_id": "llm-stats-livecodebench-pro", "caveat": "LiveCodeBench Pro is an advanced evaluation benchmark for large language models for code that uses Elo ratings to rank models based on their performance on coding tasks. It evaluates models on real-world coding problems from programming contests (LeetCode, AtCoder, CodeForces) and provides a relative ranking system where higher Elo scores indicate superior performance.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LiveCodeBench Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 605, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-pro?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-livecodebench-v5", "caveat": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LiveCodeBench v5", "organization_count": 1, "organizations": ["llm_stats"], "rank": 606, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-livecodebench-v5-24-12-25-2", "caveat": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LiveCodeBench v5 24.12-25.2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 607, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v5-24.12-25.2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-livecodebench-v6", "caveat": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LiveCodeBench v6", "organization_count": 1, "organizations": ["llm_stats"], "rank": 608, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench-v6?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-livecodebench-01-09", "caveat": "LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LiveCodeBench(01-09)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 609, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livecodebench%2801-09%29?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1787-livelongbench", "caveat": "LiveLongBench is the first spoken long-text benchmark designed to address the challenges of long-context understanding in real-world dialogues, characterized by speech-specific features, high redundancy, and uneven information density. Existing benchmarks fail to capture these complexities, limiting LiveLongBench 是首个面向口语长文本理解的基准测试，基于直播内容构建，涵盖检索类、推理类及混合类三种任务类型，针对现实对话中存在的语音特性、高冗余性和信息密度不均等挑战。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveLongBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "LiveLongBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 610, "released": "2025-04-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveLongBench"}, {"aliases": [], "benchmark_id": "opencompass-1397-livemathbench", "caveat": "LiveMathBench can capture LLM capabilities in complex reasoning tasks, including challenging latest question sets from various mathematical competitions. LiveMathBench用于评估大语言模型在复杂推理方面的表现，由极具挑战性的现代数学问题组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LiveMathBench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "LiveMathBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 611, "released": "2024-12-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LiveMathBench"}, {"aliases": [], "benchmark_id": "llm-stats-livemathematicianbench", "caveat": "LiveMathematicianBench evaluates research-level mathematical reasoning on continuously refreshed, contamination-resistant problems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "LiveMathematicianBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 612, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livemathematicianbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-livesports-3k", "caveat": "LiveSports-3K evaluates fine-grained understanding and commentary of live sports video.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "LiveSports-3K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 613, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livesports-3k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-livesqlbench", "caveat": "LiveSQLBench evaluates models on generating correct SQL queries against live PostgreSQL databases. The LiveSQLBench-Base-Full v1 dataset contains 600 questions across 22 PostgreSQL databases, testing real-world database reasoning and query generation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "LiveSQLBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 614, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/livesqlbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1360-llava-bench", "caveat": "LLaVA-Bench evaluates the model's ability in more challenging tasks, including a diverse set of 24 images with 60 questions in total, including indoor and outdoor scenes, memes, paintings, sketches, etc. LLaVA-Bench用于评估多模态大模型应对复杂任务的能力，内含24 张图像及60个问题，包括室内和室外场景、模因、绘画、素描等。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLaVA-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "LLaVA-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 615, "released": "2023-04-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LLaVA-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1838-llm-babybench", "caveat": "LLM-BabyBench is a benchmark suite designed to evaluate Large Language Models (LLMs) on grounded planning and reasoning tasks. LLM-BabyBench是一个专门评估大语言模型在交互环境中规划和推理能力的新基准测试套件。基于BabyAI网格世界的文本适配版本，该基准评估LLMs在三个核心方面的表现：预测动作对环境状态的影响（Predict任务）、生成低级动作序列以实现指定目标（Plan任务）、以及将高级指令分解为连贯的子目标序列（Decompose任务）。\n基准包含16个难度级别，提供三种文本格式（Narrative、Structured、JSON），并配备OmniBot专家代理用于生成基准数据。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLM-BabyBench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "LLM-BabyBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 616, "released": "2025-05-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LLM-BabyBench"}, {"aliases": [], "benchmark_id": "opencompass-1752-llm-srbench", "caveat": "LLM-SRBench, a comprehensive benchmark with 239 challenging problems across four scientific domains specifically designed to evaluate LLM-based scientific equation discovery methods while preventing trivial memorization. LLM-SRBench，这是一个包含239个挑战性问题的综合基准测试，涵盖四个科学领域，专门设计用于评估基于LLMs的科学方程式发现方法，同时防止简单记忆。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLM-SRBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "LLM-SRBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 617, "released": "2025-04-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LLM-SRBench"}, {"aliases": [], "benchmark_id": "opencompass-1325-llm-uncertainty-bench", "caveat": "LLM-Uncertainty-Bench is a new benchmarking approach for LLMs that integrates uncertainty quantification. It spans 5 representative natural language processing tasks, each has a dataset with 10,000 instances. LLM-Uncertainty-Bench将不确定性纳入LLM评估，包含5个具有代表性的自然语言处理任务，每个任务都有包含10000个实例的数据集支撑。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLM-Uncertainty-Bench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "LLM-Uncertainty-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 618, "released": "2024-01-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LLM-Uncertainty-Bench"}, {"aliases": [], "benchmark_id": "opencompass-2047-llmthinkbench", "caveat": "LLMThinkBench is a benchmark framework designed to evaluate large language models (LLMs) on basic math reasoning and “overthinking” behaviors, targeting code-executing language models. LLMThinkBench 是一个用于评估大语言模型（LLM）在基础数学推理和“过度思考”行为方面的基准框架，支持对具备代码执行能力的语言模型进行全面评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LLMThinkBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "LLMThinkBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 619, "released": "2025-07-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LLMThinkBench"}, {"aliases": [], "benchmark_id": "opencompass-2128-lm-evaluation-harness", "caveat": "AI Language Proficiency Monitor is a multilingual benchmark platform that systematically assesses LLM performance across up to 200 languages. AI Language Proficiency Monitor是一个多语言大语言模型评测平台，系统性评估模型在多达200种语言上的性能表现，特别关注低资源语言，整合FLORES+、MMLU、GSM8K、TruthfulQA和ARC等数据集评测翻译、问答、数学推理和事实性等能力，提供开源自动更新排行榜和交互式仪表板，覆盖全球80-95%人口使用的语言。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/lm-evaluation-harness"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "lm-evaluation-harness", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 620, "released": "2025-07-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/lm-evaluation-harness"}, {"aliases": [], "benchmark_id": "opencompass-2088-lmact", "caveat": "LMAct is a benchmark for evaluating frontier multimodal models' in-context imitation learning capabilities in long contexts. LMAct是评估前沿多模态模型在长上下文中的上下文模仿学习能力的基准。它评估了诸如井字棋、国际象棋和雅达利等交互式任务的多模式决策。该测试集包含多达100万个令牌上下文和512个专家演示集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LMAct"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "LMAct", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 621, "released": "2024-12-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LMAct"}, {"aliases": [], "benchmark_id": "llm-stats-lmarena-text", "caveat": "LMArena Text Leaderboard is a blind human preference evaluation benchmark that ranks models based on pairwise comparisons in real-world conversations. The leaderboard uses Elo ratings computed from user preferences in head-to-head model battles, providing a comprehensive measure of overall model capability and style.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LMArena Text Leaderboard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 622, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/lmarena-text?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-loca-bench-256k", "caveat": "LOCA-Bench is a long-context agentic benchmark. The 256k variant evaluates agents using the official ReAct mode with an environment description length of 256k tokens, measuring how well models reason and act over very long contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LOCA-Bench (256k)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 623, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/loca-bench-256k?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1572-loki", "caveat": "LOKI, a multimodal synthetic data detection benchmark, designed specifically to comprehensively assess the capabilities of LMMs in detecting synthetic data. LOKI是一个多模态合成数据检测基准，专门设计用于全面评估 LMMs 在检测合成数据方面的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LOKI"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "LOKI", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 624, "released": "2024-10-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LOKI"}, {"aliases": [], "benchmark_id": "opencompass-542-longbench", "caveat": "LongBench is a benchmark for bilingual, multitask, and comprehensive assessment of long context understanding capabilities of large language models. LongBench 是一个多任务、中英双语、针对大语言模型长文本理解能力的评测基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LongBench"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "LongBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 625, "released": "2023-08-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LongBench"}, {"aliases": [], "benchmark_id": "llm-stats-longbench-v2", "caveat": "LongBench v2 is a benchmark designed to assess the ability of LLMs to handle long-context problems requiring deep understanding and reasoning across real-world multitasks. It consists of 503 challenging multiple-choice questions with contexts ranging from 8k to 2M words across six major task categories: single-document QA, multi-document QA, long in-context learning, long-dialogue history understanding, code repository understanding, and long structured data understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "LongBench v2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 626, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longbench-v2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-longcodebench", "caveat": "LongCodeBench evaluates the code understanding and comprehension abilities of large language models at very long context windows, scaling up to 1M tokens. It tests whether models can reason about extensive codebases provided in a single prompt by answering multiple-choice questions about the code.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "LongCodeBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 627, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longcodebench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-longfact", "caveat": "LongFact evaluates factual precision over long-form generations containing many individual claims. Each claim is extracted and verified, and the model is scored on claim-level precision, measuring whether extended responses introduce unsupported or false statements.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500"], "document_share": 0.0008271298593879239, "domain": "factuality", "name": "LongFact", "organization_count": 1, "organizations": ["llm_stats"], "rank": 628, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-longfact-concepts", "caveat": "LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LongFact Concepts", "organization_count": 1, "organizations": ["llm_stats"], "rank": 629, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-concepts?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-longfact-objects", "caveat": "LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "LongFact Objects", "organization_count": 1, "organizations": ["llm_stats"], "rank": 630, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longfact-objects?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-longtext-bench", "caveat": "LongText-Bench evaluates text-to-image models on their ability to accurately render long text passages within generated images. It includes English (EN) and Chinese (ZH) subsets to assess multilingual text rendering capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longtext-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "image-generation", "name": "LongText-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 631, "released": "2025-07-29", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longtext-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2067-longvale", "caveat": "LongVALE is the first long video benchmark integrating fine-grained omni modalities (video, audio, speech) information in videos. It comprises 105K omni-modal events with precise temporal boundaries and detailed omni-modal captions, aiming to advance comprehensive multi-modal video understanding. LongVALE 是首个集成视频中细粒度全模态（视频、音频、语音）信息的长视频理解基准。它包含 10.5 万个具有精确时间边界和详细的全模态描述的事件标注，致力于全面提升多模态视频大语言模型的跨模态推理与细粒度时间感知能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LongVALE"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "LongVALE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 632, "released": "2025-04-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LongVALE"}, {"aliases": [], "benchmark_id": "llm-stats-longvideobench", "caveat": "LongVideoBench is a question-answering benchmark featuring video-language interleaved inputs up to an hour long. It includes 3,763 varying-length web-collected videos with subtitles across diverse themes and 6,678 human-annotated multiple-choice questions in 17 fine-grained categories for comprehensive evaluation of long-term multimodal understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "LongVideoBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 633, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/longvideobench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1510-longvideobench", "caveat": "LongVideoBench tests LMMs' understanding of long videos. It's a question-answering benchmark with video-language interleaved inputs up to an hour long and comprises 3,763 web-collected videos with subtitles across diverse themes, LongVideoBench用于评估多模态大模型的长视频理解能力，是基于交错长视频语料构建的问答集，包含3763个不同主题的带字幕的视频。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LongVideoBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "LongVideoBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 634, "released": "2024-07-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LongVideoBench"}, {"aliases": [], "benchmark_id": "opencompass-1944-loopnav", "caveat": "A video-action dataset containing many loop-based navigation dataset in Minecraft environment, aiming to boost the spatial consistency and providing insight for the design of memory module 一个视频-动作导航数据集，包括了在Minecraft环境下大量基于回环的导航数据，能够促进世界模型等视频模型空间一致性的训练，启发记忆模块的设计。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LoopNav"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "LoopNav", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 635, "released": "2025-05-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LoopNav"}, {"aliases": [], "benchmark_id": "llm-stats-lsat", "caveat": "LSAT (Law School Admission Test) benchmark evaluating complex reasoning capabilities across three challenging tasks: analytical reasoning, logical reasoning, and reading comprehension. The LSAT measures skills considered essential for success in law school including critical thinking, reading comprehension of complex texts, and analysis of arguments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "LSAT", "organization_count": 1, "organizations": ["llm_stats"], "rank": 636, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/lsat?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1335-ltmbenchmark", "caveat": "LTMbenchmark assess the long-term memory, continual learning, and information integration capabilities of the agents via dynamic conversational tasks. LTMbenchmark通过动态对话任务评估智能体的长期记忆、持续学习和信息集成能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LTMbenchmark"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "LTMbenchmark", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 637, "released": "2024-09-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LTMbenchmark"}, {"aliases": [], "benchmark_id": "opencompass-564-lv-eval", "caveat": "LV-Eval is a challenging long-context benchmark with five length levels (16k, 32k, 64k, 128k, and 256k) reaching up to 256k words. The average number of words is 102,380, and the Min/Max number of words is 11,896/387,406. It features two main tasks, single-hop QA and multi-hop QA, comprising 11 bilingual datasets. LV-Eval是一个具备5个长度等级（16k、32k、64k、128k和256k）、最大文本测试长度达到256k的长文本评测基准。LV-Eval的平均文本长度达到102,380字，最小/最大文本长度为11,896/387,406字。LV-Eval主要有两类评测任务——单跳QA和多跳QA，共包含11个涵盖中英文的评测数据子集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/LV-Eval"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "LV-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 638, "released": "2024-02-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/LV-Eval"}, {"aliases": [], "benchmark_id": "llm-stats-lvbench", "caveat": "LVBench is an extreme long video understanding benchmark designed to evaluate multimodal models on videos up to two hours in duration. It contains 6 major categories and 21 subcategories, with videos averaging five times longer than existing datasets. The benchmark addresses applications requiring comprehension of extremely long videos.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "LVBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 639, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/lvbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1147-m3t", "caveat": "M3T is a novel benchmark dataset tailored to evaluate NMT systems on the comprehensive task of translating semi-structured documents. This dataset aims to bridge the evaluation gap in document-level NMT systems, acknowledging the challenges posed by rich text layouts in real-world applications. M3T 是旨在评估神经机器翻译（NMT）系统在翻译半结构化文档的综合任务上的表现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/M3T"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "M3T", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 640, "released": "2024-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/M3T"}, {"aliases": [], "benchmark_id": "llm-stats-management-consulting-tasks", "caveat": "Management Consulting Tasks is an internal OpenAI evaluation of long-horizon professional knowledge work drawn from management-consulting workflows, scoring whether models produce correct, decision-ready analyses.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Management Consulting Tasks (Internal)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 641, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/management-consulting-tasks?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1694-maritimebench", "caveat": "MaritimeBench builds a scientific, fair maritime knowledge assessment system. With 1,888 MCQs based on industry standards, we evaluate models' capabilities across shipping domains. MaritimeBench 致力于构建一套科学、公平且严谨的航运知识评估体系。基于行业权威标准，我们持续维护并更新高质量的航运数据集——其中包含1,888道客观选择题（MCQ格式），以全面、多维度地量化模型在航运各领域的能力表现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MaritimeBench"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "MaritimeBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 642, "released": "2025-04-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MaritimeBench"}, {"aliases": [], "benchmark_id": "llm-stats-mask", "caveat": "MASK is a collection of 1000 questions measuring whether models faithfully report their beliefs when pressured to lie. It operationalizes deception as the rate at which the model lies, i.e., knowingly making false statements intended to be received as true. Lower dishonesty rates indicate better honesty.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MASK", "organization_count": 1, "organizations": ["llm_stats"], "rank": 643, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mask?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1609-mask", "caveat": "The MASK evaluation provides a rigorous benchmark for evaluating honesty in large language models by measuring whether models remain truthful when incentivized to lie. The public set contains 1,028 high-quality human-labeled examples across six distinct archetypes. MASK 评估提供了一个严格的基准，用于评估大型语言模型中的诚实度，通过测量模型在受到诱使说谎的激励时是否保持真实性。公共集包含 1,028 个高质量的人标注示例，涵盖六个不同的原型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MASK"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MASK", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 644, "released": "2025-03-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MASK"}, {"aliases": [], "benchmark_id": "opencompass-1847-massive-steps", "caveat": "Massive-STEPS is a large-scale semantic trajectories dataset designed for understanding and predicting Point-of-Interest (POI) check-ins. Massive-STEPS是一个大规模的语义轨迹数据集，旨在理解和预测兴趣点（POI）签到行为。该数据集基于Semantic Trails数据集构建，覆盖12个全球不同地区的城市，包含2012-2013年和2017-2018年的签到数据，提供了更现代和多样化的POI签到信息。Massive-STEPS不仅丰富了签到数据的语义信息，还通过与Foursquare Open Source Places数据集对齐，增加了POI的地理坐标、名称和地址等元数据。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Massive-STEPS"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "Massive-STEPS", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 645, "released": "2025-05-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Massive-STEPS"}, {"aliases": [], "benchmark_id": "opencompass-1645-mastermindeval", "caveat": "Evaluating Reasoning Capabilities of LLMs Using the Mastermind Board Game. MastermindEval使用猜谜游戏棋盘评估大型语言模型的推理能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MastermindEval"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MastermindEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 646, "released": "2025-03-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MastermindEval"}, {"aliases": [], "benchmark_id": "llm-stats-math", "caveat": "MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects including Prealgebra, Algebra, Number Theory, Counting and Probability, Geometry, Intermediate Algebra, and Precalculus.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MATH", "organization_count": 1, "organizations": ["llm_stats"], "rank": 647, "released": "2021-11-08", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/math?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-534-math", "caveat": "MATH is a new dataset of 12,500 challenging competition mathematics problems. Each problem in MATH has a full step-by-step solution. MATH 是一个包含 12,500 个具有挑战性的竞赛数学问题的新数据集。 MATH 中的每个问题都有完整的分步解决方案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MATH"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "MATH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 648, "released": "2021-11-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MATH"}, {"aliases": [], "benchmark_id": "llm-stats-math-cot", "caveat": "MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects. This variant uses Chain-of-Thought prompting to encourage step-by-step reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MATH (CoT)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 649, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/math-%28cot%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-math-500", "caveat": "MATH-500 is a subset of the MATH dataset containing 500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels across seven mathematical subjects including Prealgebra, Algebra, Number Theory, Counting and Probability, Geometry, Intermediate Algebra, and Precalculus.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MATH-500", "organization_count": 1, "organizations": ["llm_stats"], "rank": 650, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/math-500?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-matharena-apex", "caveat": "MathArena Apex is a challenging math contest benchmark featuring the most difficult mathematical problems designed to test advanced reasoning and problem-solving abilities of AI models. It focuses on olympiad-level mathematics and complex multi-step mathematical reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MathArena Apex", "organization_count": 1, "organizations": ["llm_stats"], "rank": 651, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/matharena-apex?top_n=500"}, {"aliases": ["MathArena Apex", "MathArena Apex 2025", "Apex 2025"], "benchmark_id": "matharena_apex_2025", "caveat": "The hardest MathArena split; even near-ceiling models sit well below typical math-benchmark numbers.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MathArena Apex 2025", "organization_count": 1, "organizations": ["Tencent"], "rank": 652, "released": null, "source": "model_reports", "url": "https://matharena.ai/apex"}, {"aliases": [], "benchmark_id": "opencompass-1089-mathbench", "caveat": "MathBench, a new benchmark that rigorously assesses the mathematical capabilities of large\nlanguage models. MathBench spans a wide range of mathematical disciplines, offering a\ndetailed evaluation of both theoretical understanding and practical problem-solving skills. MathBench 严格评估大型语言模型的数学能力。MathBench 涉及广泛的数学学科，提供对理论理解和实际问题解决技能的详细评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathBench"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "MathBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 653, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MathBench"}, {"aliases": [], "benchmark_id": "opencompass-1115-mathqa", "caveat": "MathQA is a new large-scale, diverse dataset of 37k English multiple-choice math word problems covering multiple math domain categories by modeling operation programs\ncorresponding to word problems in the AQuA dataset. MathQA 是一个大规模、多样化的数据集，包含 37,000 道英语多项选择数学文字问题，涵盖多个数学领域类别。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathQA"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "MathQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 654, "released": "2019-05-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MathQA"}, {"aliases": [], "benchmark_id": "llm-stats-mathverse", "caveat": "MathVerse evaluates multimodal mathematical reasoning, testing whether models genuinely interpret visual math diagrams rather than relying on text.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MathVerse", "organization_count": 1, "organizations": ["llm_stats"], "rank": 655, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1371-mathverse", "caveat": "MathVerse is intended for evaluating MLLMs' visual math problem-solving, containing 2,612 high-quality, multi-subject math problems with diagrams. MathVerse用于评估多模态大模型的视觉数学问题解决能力，包含2612个高质量、多主题的数学问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathVerse"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MathVerse", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 656, "released": "2024-03-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MathVerse"}, {"aliases": [], "benchmark_id": "llm-stats-mathverse-mini", "caveat": "MathVerse-Mini is a subset of the MathVerse benchmark for evaluating math reasoning capabilities in vision-language models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MathVerse-Mini", "organization_count": 1, "organizations": ["llm_stats"], "rank": 657, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathverse-mini?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mathvision", "caveat": "MATH-Vision is a dataset designed to measure multimodal mathematical reasoning capabilities. It focuses on evaluating how well models can solve mathematical problems that require both visual understanding and mathematical reasoning, bridging the gap between visual and mathematical domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MathVision", "organization_count": 1, "organizations": ["llm_stats"], "rank": 658, "released": "2024-02-22", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvision?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1370-mathvision", "caveat": "MathVision measures multimodal mathematical reasoning capabilities through a meticulously curated collection of 3,040 high-quality mathematical problems spanning 16 distinct mathematical disciplines and graded across 5 levels of difficulty. MathVision用于评估多模态大模型的数学推理能力，由涵盖16个数学领域、跨越5个难度级别的3040个高质量数学问题组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathVision"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MathVision", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 659, "released": "2024-02-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MathVision"}, {"aliases": [], "benchmark_id": "llm-stats-mathvista", "caveat": "MathVista evaluates mathematical reasoning of foundation models in visual contexts. It consists of 6,141 examples derived from 28 existing multimodal datasets and 3 newly created datasets (IQTest, FunctionQA, and PaperQA), combining challenges from diverse mathematical and visual tasks to assess models' ability to understand complex figures and perform rigorous reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MathVista", "organization_count": 1, "organizations": ["llm_stats"], "rank": 660, "released": "2023-10-03", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1178-mathvista", "caveat": "MathVista is a benchmark designed to combine challenges from diverse mathematical and visual tasks. It consists of 6,141 examples, derived from 28 existing multimodal datasets involving mathematics and 3 newly created datasets. MathVista用于评估多模态大模型的数学能力，结合了丰富的数学和视觉任务挑战。它由6141个示例组成，来自28个涉及数学的现有多模态数据集和3个新创建的数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MathVista"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "MathVista", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 661, "released": "2023-10-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MathVista"}, {"aliases": [], "benchmark_id": "llm-stats-mathvista-mini", "caveat": "MathVista-Mini is a smaller version of the MathVista benchmark that evaluates mathematical reasoning in visual contexts. It consists of examples derived from multimodal datasets involving mathematics, combining challenges from diverse mathematical and visual tasks to assess foundation models' ability to solve problems requiring both visual understanding and mathematical reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MathVista-Mini", "organization_count": 1, "organizations": ["llm_stats"], "rank": 662, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mathvista-mini?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-maverix", "caveat": "MAVERIX (Multimodal Audio-Visual Evaluation Reasoning Index) evaluates multimodal models on tasks that demand tight integration of video and audio information. It features challenges like situational awareness and social sentiment analysis where the answer cannot be reliably determined from a single modality, rigorously testing joint audio-visual understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MAVERIX", "organization_count": 1, "organizations": ["llm_stats"], "rank": 663, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/maverix?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1848-mavos-dd", "caveat": "MAVOS-DD is the first large-scale open-set benchmark for multilingual audio-video deepfake detection. MAVOS-DD是一个大规模的多语言音视频深度伪造检测基准数据集，包含超过250小时的真实和伪造视频，涵盖八种语言。该数据集通过七种不同的深度伪造生成模型生成伪造视频，这些模型基于不同的生成方法，包括说话头像生成、表情转移和换脸。MAVOS-DD设计了多种开放集测试场景，包括开放集模型、开放集语言和全开放集，以评估深度伪造检测模型在未知模型和语言下的泛化能力。实验结果表明，现有的深度伪造检测模型在开放集场景下的性能显著下降，凸显了开发更鲁棒检测技术的必要性。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MAVOS-DD"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MAVOS-DD", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 664, "released": "2025-05-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MAVOS-DD"}, {"aliases": [], "benchmark_id": "llm-stats-maxife", "caveat": "MAXIFE is a multilingual benchmark evaluating LLMs on instruction following and execution across multiple languages and cultural contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "MAXIFE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 665, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/maxife?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mbpp", "caveat": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MBPP", "organization_count": 1, "organizations": ["llm_stats"], "rank": 666, "released": "2021-08-16", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-538-mbpp", "caveat": "The benchmark consists of around 1,000 crowd-sourced Python programming problems, designed to be solvable by entry level programmers, covering programming fundamentals, standard library functionality, and so on. Each problem consists of a task description, code solution and 3 automated test cases. 该基准测试由大约1000个入门级程序员可以解决的众包Python编程问题组成，涵盖编程基础知识、标准库功能等。每个问题都由任务描述、代码解决方案和3个自动化测试用例组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MBPP"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "MBPP", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 667, "released": "2021-08-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MBPP"}, {"aliases": [], "benchmark_id": "llm-stats-mbpp-base-version", "caveat": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MBPP ++ base version", "organization_count": 1, "organizations": ["llm_stats"], "rank": 668, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-%2B%2B-base-version?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mbpp-evalplus", "caveat": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MBPP EvalPlus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 669, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mbpp-evalplus-base", "caveat": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MBPP EvalPlus (base)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 670, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-evalplus-%28base%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mbpp-pass-1", "caveat": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases. This variant uses pass@1 evaluation metric measuring the percentage of problems solved correctly on the first attempt.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MBPP pass@1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 671, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-pass%401?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mbpp-plus", "caveat": "MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases for more rigorous evaluation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MBPP Plus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 672, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp-plus?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mbpp-2", "caveat": "MBPP+ is an enhanced version of MBPP (Mostly Basic Python Problems) with significantly more test cases (35x) for more rigorous evaluation. MBPP is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MBPP+", "organization_count": 1, "organizations": ["llm_stats"], "rank": 673, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mbpp%2B?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1606-mcitebench", "caveat": "MCiteBench is a benchmark to evaluate multimodal citation text generation in MLLMs. It consists of 3,000 samples from 1,749 academic papers, featuring 2,000 Explanation tasks and 1,000 Locating tasks, with balanced evidence across text, figures, tables, and mixed modalities. MCiteBench 是一个用于评估多模态大模型中多模态引用文本生成的基准，由 1,749 篇学术论文中的 3,000 个样本组成，包括 2,000 个解释任务和 1,000 个定位任务，在文本、图表、表格和混合模态中平衡证据。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MCiteBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MCiteBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 674, "released": "2025-03-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MCiteBench"}, {"aliases": [], "benchmark_id": "llm-stats-mcp-atlas", "caveat": "MCP Atlas is a benchmark for evaluating AI models on scaled tool use capabilities, measuring how well models can coordinate and utilize multiple tools across complex multi-step tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MCP Atlas", "organization_count": 1, "organizations": ["llm_stats"], "rank": 675, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-atlas?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mcp-mark", "caveat": "MCP-Mark evaluates LLMs on their ability to use Model Context Protocol (MCP) tools effectively, testing tool discovery, selection, invocation, and result interpretation across diverse MCP server scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "MCP-Mark", "organization_count": 1, "organizations": ["llm_stats"], "rank": 676, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-mark?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mcp-universe", "caveat": "MCP-Universe evaluates LLMs on complex multi-step agentic tasks using Model Context Protocol (MCP) tools across diverse interactive environments, testing planning, tool orchestration, and task completion.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "MCP-Universe", "organization_count": 1, "organizations": ["llm_stats"], "rank": 677, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mcp-universe?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-measurebench", "caveat": "MeasureBench evaluates multimodal models on visual measurement and quantitative perception tasks across both real and synthetic imagery, reported as the average over the two settings.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MeasureBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 678, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/measurebench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1641-medagents-bench", "caveat": "MedAgentsBench features 862 challenging medical questions from seven datasets, focusing on cases where models struggle. It emphasizes multi-step clinical reasoning and addresses limitations in existing benchmarks by eliminating simple questions and standardizing evaluation protocols. MedAgentsBench是一个专注于复杂医学推理的基准测试，从七个医学数据集中精选了862个挑战性问题。这些数据集包括MedQA、PubMedQA、MedMCQA、MedBullets、MedExQA、MedXpertQA和MMLU/MMLU-Pro，涵盖了从医学执照考试到研究文献的多种医学问题。\n该基准选择少于50%模型能正确回答的问题，确保医学知识领域全面覆盖，并优先选择需要多步临床推理的问题。这解决了现有评估中简单问题普遍存在、评估协议不一致，以及缺乏性能-成本-时间分析的局限。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedAgents-Bench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MedAgents-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 679, "released": "2025-03-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedAgents-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1893-medal", "caveat": "MEDAL is a framework for generating and evaluating multilingual open-domain chatbots and their evaluators. This framework supports various language models and provides a structured approach to creating conversational datasets. MEDAL是一个用于生成和评估多语言开放域聊天机器人及其评估器的框架。该框架支持各种语言模型,并提供了一种结构化的方法来创建对话数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MEDAL"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MEDAL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 680, "released": "2025-05-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MEDAL"}, {"aliases": [], "benchmark_id": "opencompass-1874-medarabiq", "caveat": "MedArabiQ introduces a new benchmark dataset consisting of seven Arabic medical tasks, covering multiple specialties and question formats:\n✅ Multiple-choice questions\n✏️ Fill-in-the-blank (with and without choices)\n💬 Patient-doctor question answering 大型语言模型（LLMs）在医疗保健应用中显示出了显著的前景，但由于缺乏高质量的领域特定数据集，它们在阿拉伯语医学领域的表现在很大程度上仍未得到探索。MedArabiQ引入了一个新的基准数据集，包含七个阿拉伯语医学任务，涵盖多个专业和问题格式：\n✅ 多项选择题\n✏️ 填空题（有选项和无选项）\n💬 患者-医生问答\n\n该数据集使用过往医学考试和公开可用资源构建，并进行了修改以评估LLMs在各种能力方面的表现，包括偏见缓解。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedArabiQ"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MedArabiQ", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 681, "released": "2025-05-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedArabiQ"}, {"aliases": [], "benchmark_id": "opencompass-1287-medbench", "caveat": "MedBench is committed to building a scientific, fair and rigorous Chinese medical model evaluation system and open platform. Based on authoritative standards, we constantly update and maintain high-quality datasets, and comprehensively quantify capabilities of models in various medical dimensions. MedBench致力于打造一个科学、公平且严谨的中文医疗大模型评测体系及开放平台。我们基于医学权威标准，不断更新维护高质量的医学数据集，全方位多维度量化模型在各个医学维度的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedBench"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "MedBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 682, "released": "2023-12-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedBench"}, {"aliases": [], "benchmark_id": "opencompass-1920-medbookvqa", "caveat": "MedBookVQA is a medical visual question answering (VQA) benchmark constructed from open-access medical textbooks. It includes 5,000 questions across five clinical task types and is hierarchically organized by imaging modality, anatomical structure, and clinical specialty. MedBookVQA 是一个基于开放获取医学教科书构建的医学视觉问答（VQA）基准数据集。它包含 5,000 个问题，涵盖五种临床任务类型，并按照影像模态、解剖结构和临床专科进行分层组织。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedBookVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MedBookVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 683, "released": "2025-05-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedBookVQA"}, {"aliases": [], "benchmark_id": "opencompass-1832-medbrowsecomp", "caveat": "the first benchmark that systematically tests an agent’s ability to reliably retrieve and synthesize multi-hop medical facts from live, domain-specific knowledge bases. MedBrowseComp holds 1,000+ human-curated questions that mirror clinical scenarios\n﻿ the first benchmark that systematically tests an agent’s ability to reliably retrieve and synthesize multi-hop medical facts from live, domain-specific knowledge bases. MedBrowseComp holds 1,000+ human-curated questions that mirror clinical scenarios\n﻿", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedBrowseComp"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MedBrowseComp", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 684, "released": "2025-05-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedBrowseComp"}, {"aliases": [], "benchmark_id": "opencompass-1241-medcalc-bench", "caveat": "MedCalc-Bench focuses on evaluating the medical calculation capability of LLMs. It contains an evaluation set of over 1000 manually reviewed instances from 55 different medical calculation tasks. MedCalc-Bench专注于评估LLM的医学计算能力，包含来自55个不同医学计算任务的1000个经过人工审查的实例。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedCalc-Bench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MedCalc-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 685, "released": "2024-06-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedCalc-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-medchembench", "caveat": "MedChemBench is an internal OpenAI evaluation of medicinal-chemistry reasoning, testing whether models can support drug-discovery-relevant chemistry analysis and decision-making.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MedChemBench (Internal)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 686, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/medchembench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2417-medhalltune", "caveat": null, "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedHallTune"], "document_share": 0.0008271298593879239, "domain": "医学", "name": "MedHallTune", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 687, "released": "2025-02-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedHallTune"}, {"aliases": [], "benchmark_id": "opencompass-1548-medhallu", "caveat": "MedHallu is a comprehensive benchmark dataset designed to evaluate the ability of large language models to detect hallucinations in medical question-answering tasks. MedHallu 旨在评估大型语言模型在医学问题解答任务中检测幻觉的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedHallu"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MedHallu", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 688, "released": "2025-02-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedHallu"}, {"aliases": [], "benchmark_id": "opencompass-1326-medjourney", "caveat": "MedJourney offers a comprehensive assessment of LLMs' effectiveness in real-world clinical settings. It includes multiple tasks from 4 stages of a typical patient's hospital visit journey and comprises 12 datasets. MedJourney用于评估 LLM 在真实临床环境中的有效性，其中包含多个任务，涵盖来自患者就诊典型流程的4个阶段的12个数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedJourney"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MedJourney", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 689, "released": "2024-09-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedJourney"}, {"aliases": [], "benchmark_id": "opencompass-1331-medsafetybench", "caveat": "MedSafetyBench is designed to measure the medical safety of LLMs. It includes 1,800 medical safety demonstrations, where each safety demonstration consists of a harmful medical request and a corresponding safe response. MedSafetyBench用于评估LLM在医疗安全上的表现，包含1800个由有害请求和安全响应组成的医疗安全场景。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedSafetyBench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "MedSafetyBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 690, "released": "2024-03-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedSafetyBench"}, {"aliases": [], "benchmark_id": "llm-stats-medxpertqa", "caveat": "A comprehensive benchmark to evaluate expert-level medical knowledge and advanced reasoning, featuring 4,460 questions spanning 17 specialties and 11 body systems. Includes both text-only and multimodal subsets with expert-level exam questions incorporating diverse medical images and rich clinical information.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MedXpertQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 691, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1891-medxpertqa", "caveat": "We introduce MedXpertQA, a highly challenging and comprehensive medical benchmark to evaluate expert-level medical knowledge and advanced reasoning. \nIt has been accepted by ICML 2025 and selected by Google DeepMind as the benchmark for MedGemma. MedXpertQA是由清华大学和上海人工智能实验室构建的全面且具有高度挑战性的医学基准，用于评估专家级的医学知识和高级推理能力。论文已被ICML 2025接收，并被Google DeepMind使用作为MedGemma的评估基准。MedXpertQA 共包含 4,460 道题目，涵盖 17 个医学专科和 11 个身体系统。该基准包含两个子集：用于文本医学能力评估的 Text 子集，以及用于多模态医学能力评估的 MM 子集。MM 子集首次引入了带有多样化图像和丰富临床信息（如病历和检查结果）的专家级考试题，区别于传统多模态医学基准中基于图像描述生成的简单问答对。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MedXpertQA"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "MedXpertQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 692, "released": "2025-02-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MedXpertQA"}, {"aliases": [], "benchmark_id": "llm-stats-medxpertqa-mm", "caveat": "MedXpertQA-MM is the multimodal subset of MedXpertQA, evaluating expert-level medical question answering grounded in medical images.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500"], "document_share": 0.0008271298593879239, "domain": "medical", "name": "MedXpertQA-MM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 693, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/medxpertqa-mm?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mega-mlqa", "caveat": "MLQA as part of the MEGA (Multilingual Evaluation of Generative AI) benchmark suite. A multi-way aligned extractive QA evaluation benchmark for cross-lingual question answering across 7 languages (English, Arabic, German, Spanish, Hindi, Vietnamese, and Simplified Chinese) with over 12K QA instances in English and 5K in each other language.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MEGA MLQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 694, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-mlqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mega-tydi-qa", "caveat": "TyDi QA as part of the MEGA benchmark suite. A question answering dataset covering 11 typologically diverse languages (Arabic, Bengali, English, Finnish, Indonesian, Japanese, Korean, Russian, Swahili, Telugu, and Thai) with 204K question-answer pairs. Features realistic information-seeking questions written by people who want to know the answer but don't know it yet.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MEGA TyDi QA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 695, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-tydi-qa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mega-udpos", "caveat": "Universal Dependencies POS tagging as part of the MEGA benchmark suite. A multilingual part-of-speech tagging dataset based on Universal Dependencies treebanks, utilizing the universal POS tag set of 17 tags across 38 diverse languages from different language families. Used for evaluating multilingual POS tagging systems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "MEGA UDPOS", "organization_count": 1, "organizations": ["llm_stats"], "rank": 696, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-udpos?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mega-xcopa", "caveat": "XCOPA (Cross-lingual Choice of Plausible Alternatives) as part of the MEGA benchmark suite. A typologically diverse multilingual dataset for causal commonsense reasoning in 11 languages, including resource-poor languages like Eastern Apurímac Quechua and Haitian Creole. Requires models to select which choice is the effect or cause of a given premise.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MEGA XCOPA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 697, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xcopa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mega-xstorycloze", "caveat": "XStoryCloze as part of the MEGA benchmark suite. A cross-lingual story completion task that consists of professionally translated versions of the English StoryCloze dataset to 10 non-English languages. Requires models to predict the correct ending for a given four-sentence story, evaluating commonsense reasoning and narrative understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MEGA XStoryCloze", "organization_count": 1, "organizations": ["llm_stats"], "rank": 698, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mega-xstorycloze?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-meld", "caveat": "MELD (Multimodal EmotionLines Dataset) is a multimodal multi-party dataset for emotion recognition in conversations. Contains approximately 13,000 utterances from 1,433 dialogues extracted from the TV series Friends. Each utterance is annotated with emotion (Anger, Disgust, Sadness, Joy, Neutral, Surprise, Fear) and sentiment labels across audio, visual, and textual modalities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Meld", "organization_count": 1, "organizations": ["llm_stats"], "rank": 699, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/meld?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2079-mer-unibench", "caveat": "MER-UniBench is a benchmark designed to evaluate multimodal large language models (MLLMs) for their emotion understanding capabilities across typical multimodal emotion recognition (MER) tasks. MER-UniBench 是一个旨在评估多模态大语言模型（MLLM）在典型多模态情感识别（MER）任务中情感理解能力的评测基准。它涵盖了细粒度情感识别、基本情感识别和情感分析三个主要维度。该基准利用了包括 MER-Caption 在内的多个数据集，其中 MER-Caption 拥有超过 2000 种细粒度情感类别和 11.5 万个样本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MER-UniBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MER-UniBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 700, "released": "2025-01-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MER-UniBench"}, {"aliases": [], "benchmark_id": "llm-stats-meta-internal-coding-bench", "caveat": "Meta's internal evaluation of coding-agent performance.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Meta Internal Coding Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 701, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/meta-internal-coding-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mewc", "caveat": "MEWC is a benchmark that evaluates AI model performance on multi-environment web challenges, testing agents' ability to navigate and complete complex tasks across diverse web environments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MEWC", "organization_count": 1, "organizations": ["llm_stats"], "rank": 702, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mewc?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mgsm", "caveat": "MGSM (Multilingual Grade School Math) is a benchmark of grade-school math problems. Contains 250 grade-school math problems manually translated from the GSM8K dataset into ten typologically diverse languages: Spanish, French, German, Russian, Chinese, Japanese, Thai, Swahili, Bengali, and Telugu. Evaluates multilingual mathematical reasoning capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MGSM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 703, "released": "2022-10-06", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mgsm?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-miabench", "caveat": "MIABench evaluates multimodal instruction alignment and following capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MIABench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 704, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/miabench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2085-mib", "caveat": "MIB is a benchmark designed to evaluate mechanistic interpretability methods for neural language models. MIB 是一个旨在评估神经网络语言模型中机械可解释性方法的基准。它主要评估方法在精确地定位因果路径（电路定位）和特定概念（因果变量定位）方面的能力，侧重于忠实度和最小化电路规模。该基准涵盖了IOI和算术等四个任务，并评估了Llama-3.1和Gemma-2等四种模型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIB"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "MIB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 705, "released": "2025-04-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MIB"}, {"aliases": [], "benchmark_id": "opencompass-1667-microvqa", "caveat": "MicroVQA is a benchmark that evaluates LLM reasoning on multiple-choice questions about microscopy images, created by expert biologists. MicroVQA，一个评估关于显微镜图像的多选题推理基准，由专家生物学家创建，旨在反映生物研究中能够有意义地协助的任务，每个问题都需要多模态推理。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MicroVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MicroVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 706, "released": "2025-03-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MicroVQA"}, {"aliases": [], "benchmark_id": "opencompass-1781-mieb", "caveat": "Massive Image Embedding Benchmark (MIEB) is to evaluate the performance of image and image-text embedding models across the broadest spectrum to date. MIEB用于评估图像和图像文本嵌入模型在迄今为止最广泛的范围内的性能。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIEB"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MIEB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 707, "released": "2025-04-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MIEB"}, {"aliases": [], "benchmark_id": "opencompass-1658-milic-eval", "caveat": "MiLiC-Eval is an NLP evaluation suite for Minority Languages in China, covering Tibetan (bo), Uyghur (ug), Kazakh (kk, in the Kazakh Arabic script), and Mongolian (mn, in the traditional Mongolian script). MiLiC-Eval 是针对中国少数民族语言的 NLP 评估套件，涵盖藏语（bo）、维吾尔语（ug）、哈萨克语（kk，使用哈萨克阿拉伯文脚本）和蒙古语（mn，使用传统蒙古文脚本）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MiLiC-Eval"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MiLiC-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 708, "released": "2025-03-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MiLiC-Eval"}, {"aliases": [], "benchmark_id": "llm-stats-mimic-cxr", "caveat": "MIMIC-CXR is a large publicly available dataset of chest radiographs with free-text radiology reports. Contains 377,110 images corresponding to 227,835 radiographic studies from 65,379 patients at Beth Israel Deaconess Medical Center. The dataset is de-identified and widely used for medical imaging research, automated report generation, and medical AI development.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MIMIC CXR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 709, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mimic-cxr?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mimo-coding-bench", "caveat": "MiMo Coding Bench evaluates coding-agent capabilities on software engineering tasks reported with the MiMo model family.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "MiMo Coding Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 710, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mimo-coding-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1125-mind2web", "caveat": "MIND2WEB is the first dataset for developing and evaluating generalist agents for the web that can follow language instructions to complete complex tasks on any website. MIND2WEB 是首个用于开发和评估通用网页代理的数据集，能够根据语言指令在任何网站上完成复杂任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Mind2Web"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "Mind2Web", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 711, "released": "2023-12-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Mind2Web"}, {"aliases": [], "benchmark_id": "llm-stats-minerva", "caveat": "Minerva is a benchmark for complex video reasoning, evaluating models on multi-step reasoning over long and information-dense video content.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Minerva", "organization_count": 1, "organizations": ["llm_stats"], "rank": 712, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/minerva?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1518-minictx", "caveat": "A context-rich benchmark for evaluating neural theorem proving in realistic scenarios, providing premises, full context, multi-source benchmark, and temporal splits, enabling evaluation of a model's ability to work with context that evolves over time. 一个用于评估现实场景中神经定理证明的丰富上下文基准。通过提供前提、完整上下文、多源基准和时序分割，突出其评估模型处理随时间演变的上下文能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/miniCTX"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "miniCTX", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 713, "released": "2024-08-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/miniCTX"}, {"aliases": [], "benchmark_id": "opencompass-1853-minilongbench", "caveat": "MiniLongBench is a low-cost benchmark for evaluating the Long Context Understanding (LCU) capabilities of LLMs, featuring a compact yet diverse test set of only 237 samples spanning 6 major task categories and 21 distinct tasks. MiniLongBench is a low-cost benchmark for evaluating the Long Context Understanding (LCU) capabilities of LLMs, featuring a compact yet diverse test set of only 237 samples spanning 6 major task categories and 21 distinct tasks.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MiniLongBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "MiniLongBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 714, "released": "2025-05-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MiniLongBench"}, {"aliases": [], "benchmark_id": "opencompass-1844-miracl-vision", "caveat": "MIRACL-VISION is a large-scale, multilingual visual document retrieval benchmark built by the NVIDIA team, extending the popular MIRACL multilingual text retrieval benchmark. It covers 18 languages and contains 211 original questions. MIRACL-VISION是一个大规模的多语言视觉文档检索基准测试，由NVIDIA团队构建，扩展了流行的MIRACL多语言文本检索基准。该数据集覆盖18种语言，包含211个原创问题，涵盖边界层分析、WKB方法、非线性偏微分方程的渐近解和振荡积分的渐近性等核心主题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIRACL-VISION"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MIRACL-VISION", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 715, "released": "2025-05-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MIRACL-VISION"}, {"aliases": [], "benchmark_id": "opencompass-1142-mirage", "caveat": "MIRAGE is a first-of-its-kind benchmark including 7,663 questions from five medical QA datasets. MIRAGE 是首个此类基准，包含来自五个医学问答数据集的 7,663 个问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MIRAGE"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MIRAGE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 716, "released": "2024-08-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MIRAGE"}, {"aliases": [], "benchmark_id": "opencompass-1623-mj-bench", "caveat": "MJ-Bench incorporates a comprehensive preference dataset to evaluate multimodal judges in providing feedback for image generation models across four key perspectives: alignment, safety, image quality, and bias. MJ-Bench，它包含了一个综合的偏好数据集，用于从四个关键角度评估多模态评委在为图像生成模型提供反馈方面的能力：对齐、安全性、图像质量和偏见。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MJ-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MJ-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 717, "released": "2024-07-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MJ-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1099-mkqa", "caveat": "MKQA is an open-domain question answering evaluation set comprising 10k question-answer pairs aligned across 26 typologically diverse languages (260k question-answer pairs in total). MKQA 是一个开放域问答评估集，包含 10,000 对问题和答案，涵盖 26 种类型多样的语言（总计 260,000 对问题和答案）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MKQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "MKQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 718, "released": "2021-08-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MKQA"}, {"aliases": [], "benchmark_id": "artificial-analysis-mlcr-aa", "caveat": "MLCR-AA overall score", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/mlcr-aa"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MLCR-AA", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 719, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/mlcr-aa"}, {"aliases": [], "benchmark_id": "llm-stats-mle-bench", "caveat": "MLE-Bench evaluates AI agents on machine learning engineering tasks by measuring their performance on Kaggle competitions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "MLE-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 720, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mle-bench-lite", "caveat": "MLE-Bench Lite evaluates AI agents on machine learning engineering tasks, testing their ability to build, train, and optimize ML models for Kaggle-style competitions in a lightweight evaluation format.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "MLE-Bench Lite", "organization_count": 1, "organizations": ["llm_stats"], "rank": 721, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mle-bench-lite?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1779-mlrc-bench", "caveat": "MLRC-Bench, a benchmark designed to quantify how effectively language agents can tackle challenging Machine Learning (ML) Research Competitions. MLRC-Bench旨在量化大模型代理如何有效地应对具有挑战性的机器学习 （ML） 研究竞赛。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MLRC-Bench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "MLRC-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 722, "released": "2025-04-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MLRC-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-mls-bench-lite", "caveat": "MLS-Bench Lite is the official 30-task subset of MLS-Bench for evaluating whether AI systems can invent generalizable and scalable machine learning methods across LLM pretraining and post-training, robotics, world models, computer vision, reinforcement learning, optimization, ML systems, and AI for Science.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MLS-Bench Lite", "organization_count": 1, "organizations": ["llm_stats"], "rank": 723, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mls-bench-lite?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mlvu", "caveat": "A comprehensive benchmark for multi-task long video understanding that evaluates multimodal large language models on videos ranging from 3 minutes to 2 hours across 9 distinct tasks including reasoning, captioning, recognition, and summarization.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MLVU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 724, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1512-mlvu", "caveat": "MLVU test MLLMs' understanding of multi-task long videos, encompassing diverse tasks based on various long videos. MLVU用于评估多模态大模型的长视频理解能力，包含面向各种类型长视频的多样化的评估任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MLVU"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MLVU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 725, "released": "2024-06-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MLVU"}, {"aliases": [], "benchmark_id": "llm-stats-mlvu-m", "caveat": "MLVU-M benchmark", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "MLVU-M", "organization_count": 1, "organizations": ["llm_stats"], "rank": 726, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mlvu-m?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mm-if-eval", "caveat": "A challenging multimodal instruction-following benchmark that includes both compose-level constraints for output responses and perception-level constraints tied to input images, with comprehensive evaluation pipeline.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MM IF-Eval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 727, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-if-eval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1558-mm-alignbench", "caveat": "A benchmark for evaluating MLLMs' alignment with human preferences. It includes 252 high-quality, human-annotated samples with diverse image types and open-ended questions. Modeled after Arena-style benchmarks, it uses GPT-4o as the judge model and Claude-Sonnet-3 as the reference model. 用于评估 MLLM 与人类偏好的一致性的基准。它包含 252 个高质量、人类标注的样本，具有不同的图像类型和开放式问题。它仿照 Arena 风格的基准，使用 GPT-4o 作为评判模型，Claude-Sonnet-3 作为参考模型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-AlignBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MM-AlignBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 728, "released": "2025-02-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-AlignBench"}, {"aliases": [], "benchmark_id": "llm-stats-mm-browsercomp", "caveat": "MM-BrowserComp evaluates multimodal agents on web browsing and information retrieval tasks, testing a model's ability to perceive, navigate, and extract information from real web environments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MM-BrowserComp", "organization_count": 1, "organizations": ["llm_stats"], "rank": 729, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-browsercomp?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mm-clawbench", "caveat": "MM-ClawBench evaluates models on MiniMax's Claw-style agent benchmark, measuring practical agentic task completion quality in real-world OpenClaw usage scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "MM-ClawBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 730, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-clawbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1566-mm-iq", "caveat": "MM-IQ is a comprehensive evaluation framework comprising 2,710 meticulously curated test items spanning 8 distinct reasoning paradigms. MM-IQ，这是一个包含 2,710 个精心挑选的测试项目的综合评估框架，涵盖了 8 种不同的推理范式。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-IQ"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MM-IQ", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 731, "released": "2025-02-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-IQ"}, {"aliases": [], "benchmark_id": "llm-stats-mm-mind2web", "caveat": "A multimodal web navigation benchmark comprising 2,000 open-ended tasks spanning 137 websites across 31 domains. Each task includes HTML documents paired with webpage screenshots, action sequences, and complex web interactions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MM-Mind2Web", "organization_count": 1, "organizations": ["llm_stats"], "rank": 732, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mind2web?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mm-mt-bench", "caveat": "A multi-turn LLM-as-a-judge evaluation benchmark for testing multimodal instruction-tuned models' ability to follow user instructions in multi-turn dialogues and answer open-ended questions in a zero-shot manner.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MM-MT-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 733, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mm-mt-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1539-mm-rlhf", "caveat": "MM-RLHF, a comprehensive project for aligning Multimodal Large Language Models (MLLMs) with human preferences. The dataset and algorithms enable consistent performance improvements across 10 dimensions and 27 benchmarks for open-source MLLMs. MM-RLHF，这是一个将多模态大型语言模型（MLLMs）与人类偏好对齐的全面项目，使开源多语言机器学习模型在 10 个维度和 27 个基准测试中实现持续的性能提升。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-RLHF"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MM-RLHF", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 734, "released": "2025-02-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-RLHF"}, {"aliases": [], "benchmark_id": "opencompass-1356-mm-vet", "caveat": "MM-Vet evaluates the capabilities to deal with complicated multimodal tasks, defining 6 core VL capabilities and examining the 16 integrations of interest derived from the capability combination. MM-Vet用于评估复杂多模态任务能力，涵盖了6个核心视觉语言功能的16种功能组合。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MM-Vet"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MM-Vet", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 735, "released": "2023-08-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MM-Vet"}, {"aliases": [], "benchmark_id": "opencompass-1585-mmad", "caveat": "MMAD is the first-ever full-spectrum MLLMs benchmark in industrial Anomaly Detection. \nResearchers defined seven key subtasks of MLLMs in industrial inspection and designed a novel pipeline to generate the MMAD dataset with 39,672 questions for 8,366 industrial images. MMAD是第一个工业异常检测领域的全谱 MLLMs 基准，研究人员定义了工业检测中 MLLMs 的七个关键子任务，并设计了一个新颖的流程来生成包含 39,672 个问题以及 8,366 个工业图像的 MMAD 数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMAD"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMAD", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 736, "released": "2024-10-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMAD"}, {"aliases": [], "benchmark_id": "opencompass-1910-mmar", "caveat": "We introduce MMAR, a new benchmark comprising 1,000 meticulously curated audio-question-answer triplets, designed to evaluate the deep reasoning capabilities of Audio-Language Models (ALMs) across massive multi-disciplinary tasks. MMAR是一个全新评测基准，旨在评估音频-语言模型（ALMs）的深度推理能力。该基准包含1,000个精心构建的音频与问答，并经过多轮纠错与质量校验以确保高标准。与现有局限于特定声音、音乐或语音领域的评测体系不同，MMAR覆盖现实场景中的混合模态，并采用四级分层分类体系（信号层、感知层、语义层与文化层）。该基准中的每个题目都需要超越表层理解的多层次深度推理，部分问题更要求研究生级别的专业领域知识与感知能力。我们在MMAR上评估了多类模型，测试结果表明该基准具有显著挑战性，分析结果进一步揭示了当前模型在理解与推理能力上的关键局限。我们期待MMAR能推动这个重要但尚未充分探索的研究领域的发展。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMAR"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "MMAR", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 737, "released": "2025-05-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMAR"}, {"aliases": [], "benchmark_id": "llm-stats-mmau", "caveat": "A massive multi-task audio understanding and reasoning benchmark comprising 10,000 carefully curated audio clips paired with human-annotated natural language questions spanning speech, environmental sounds, and music. Requires expert-level knowledge and complex reasoning across 27 distinct skills.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMAU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 738, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmau-music", "caveat": "A subset of the MMAU benchmark focused specifically on music understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across music audio clips.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMAU Music", "organization_count": 1, "organizations": ["llm_stats"], "rank": 739, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-music?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmau-sound", "caveat": "A subset of the MMAU benchmark focused specifically on environmental sound understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across environmental sound clips.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMAU Sound", "organization_count": 1, "organizations": ["llm_stats"], "rank": 740, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-sound?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmau-speech", "caveat": "A subset of the MMAU benchmark focused specifically on speech understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across speech audio clips.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMAU Speech", "organization_count": 1, "organizations": ["llm_stats"], "rank": 741, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmau-speech?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmbc", "caveat": "MMBC is a multimodal benchmark for vision-language knowledge and reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMBC", "organization_count": 1, "organizations": ["llm_stats"], "rank": 742, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbc?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmbench", "caveat": "A bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks with robust metrics.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 743, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1206-mmbench", "caveat": "MMBench is a collection of benchmarks to evaluate the multi-modal understanding capability of large vision language models (LVLMs). This benchmark contains 3,000 multiple-choice questions covering 20 fine-grained assessment dimensions. MMBench是OpenCompass 研究团队自建的视觉语言模型评测数据集，可实现从感知到认知能力逐级细分评估。此评测基准包含3000 道单项选择题 ，覆盖 20个细粒度评估维度。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 744, "released": "2023-07-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMBench"}, {"aliases": [], "benchmark_id": "llm-stats-mmbench-v1-1", "caveat": "Version 1.1 of MMBench, an improved bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMBench-V1.1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 745, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-v1.1?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmbench-video", "caveat": "A long-form multi-shot benchmark for holistic video understanding that incorporates approximately 600 web videos from YouTube spanning 16 major categories, with each video ranging from 30 seconds to 6 minutes. Includes roughly 2,000 original question-answer pairs covering 26 fine-grained capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMBench-Video", "organization_count": 1, "organizations": ["llm_stats"], "rank": 746, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmbench-video?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1207-mmbench-video", "caveat": "MMBench-Video is a comprehensive video understanding evaluation benchmark that covers long videos, multiple shots, and evaluates the temporal understanding ability of MLLMs. Contains over 600 videos, 16 categories, and manually annotated Q&A pairs. MMBench-Video是全面视频理解评测基准，覆盖长视频、多镜头，评估MLLMs时序理解能力。包含16类共600+视频以及人工标注问答对。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMBench-Video"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMBench-Video", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 747, "released": "2024-06-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMBench-Video"}, {"aliases": [], "benchmark_id": "opencompass-1334-mmdu", "caveat": "MMDU is intended for evaluating the multi-image multi-turn dialogue capabilities. It comprises 110 high-quality multi-image multi-turn dialogues with more than 1600 questions, each accompanied by detailed long-form answers. MMDU用于评估大型视觉语言模型的多图像多轮对话能力，包含110个高质量的多图像多轮对话，由1600多个附有详细的长篇答案的问题组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMDU"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMDU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 748, "released": "2024-06-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMDU"}, {"aliases": [], "benchmark_id": "llm-stats-mme", "caveat": "A comprehensive evaluation benchmark for Multimodal Large Language Models measuring both perception and cognition abilities across 14 subtasks. Features manually designed instruction-answer pairs to avoid data leakage and provides systematic quantitative assessment of MLLM capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MME", "organization_count": 1, "organizations": ["llm_stats"], "rank": 749, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mme?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1357-mme", "caveat": "MME is a comprehensive MLLM evaluation benchmark. It measures both perception and cognition abilities on a total of 14 subtasks. MME是一个全面的多模态大模型评估基准，涵盖14 个考察感知和认知能力的子任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MME"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MME", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 750, "released": "2023-06-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MME"}, {"aliases": [], "benchmark_id": "opencompass-1565-mme-cot", "caveat": "MME-CoT is a specialized benchmark evaluating the CoT reasoning performance of LMMs, spanning six domains: math, science, OCR, logic, space-time, and general scenes. MME-CoT，一个专门用于评估 LMMs CoT 推理性能的基准，涵盖六个领域：数学、科学、OCR、逻辑、时空和一般场景。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MME-CoT"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "MME-CoT", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 751, "released": "2025-02-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MME-CoT"}, {"aliases": [], "benchmark_id": "llm-stats-mme-realworld", "caveat": "A comprehensive evaluation benchmark for Multimodal Large Language Models featuring over 13,366 high-resolution images and 29,429 question-answer pairs across 43 subtasks and 5 real-world scenarios. The largest manually annotated multimodal benchmark to date, designed to test MLLMs on challenging high-resolution real-world scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MME-RealWorld", "organization_count": 1, "organizations": ["llm_stats"], "rank": 752, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mme-realworld?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1451-mme-realworld", "caveat": "MME-RealWorld evaluates MLLMs' real-world recognition, featuring 13,366 high-resolution images averaging 2,000 × 1,500 pixels. MME-RealWorld用于评估多模态大模型对真实场景的理解能力，包含13366个平均2000*1500像素的高分辨率图像。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MME-RealWorld"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MME-RealWorld", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 753, "released": "2024-08-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MME-RealWorld"}, {"aliases": [], "benchmark_id": "opencompass-1523-mmie", "caveat": "MMIE, a large-scale knowledge-intensive benchmark for evaluating interleaved multimodal comprehension and generation in Large Vision-Language Models (LVLMs). MMIE，这是一个大规模知识密集型基准，用于评估大型视觉-语言模型（LVLMs）中的交错多模态理解和生成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMIE"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMIE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 754, "released": "2024-10-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMIE"}, {"aliases": [], "benchmark_id": "opencompass-1547-mmir", "caveat": "A benchmark for evaluating Multimodal Large Language Models (MLLMs) on detecting and reasoning about inconsistencies in layout-rich multimodal content. MMIR features 534 challenging samples across five reasoning-heavy inconsistency categories. 用于评估多模态模型（MLLM）检测和推理布局丰富的多模态内容中的不一致性的基准。MMIR 包含 534 个具有挑战性的样本，涉及五个推理能力较强的不一致类别。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMIR"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMIR", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 755, "released": "2025-02-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMIR"}, {"aliases": [], "benchmark_id": "opencompass-1453-mmiu", "caveat": "MMIU test MLLMs' multi-image understanding capabilities, encompassesing 7 types of multi-image relationships, 52 tasks, 77K images, and 11K meticulously curated multiple-choice questions. MMIU用于评估多模态大模型的多图理解能力，包含7种类型的多图像关系、52个任务、77K图像和11K精心策划的多选题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMIU"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMIU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 756, "released": "2024-08-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMIU"}, {"aliases": [], "benchmark_id": "opencompass-1578-mmke-bench", "caveat": "MMKE-Bench is a benchmark designed to evaluate the ability of LMMs to edit visual knowledge in real-world scenarios. It includes 2,940 pieces of knowledge and 8,363 images across 33 broad categories, with automatically generated, human-verified evaluation questions. MMKE-Bench是一个旨在评估 LMM 在现实场景中编辑视觉知识能力的基准，包括 33 个广泛类别中的 2,940 条知识和 8,363 张图像，以及自动生成并由人工验证的评估问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMKE-Bench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MMKE-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 757, "released": "2025-02-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMKE-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1851-mmlongbench", "caveat": "MMLongBench is a benchmark that evaluates long-context vision-language models across various tasks, image types, and input lengths, revealing that single-task performance is insufficient for gauging overall vision-language long-context capability. MMLongBench 是一个针对长上下文视觉-语言模型的基准，覆盖多种任务、图像类型和输入长度。评测结果表明，单一任务的表现不足以衡量模型的整体长上下文能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLongBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMLongBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 758, "released": "2025-05-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench"}, {"aliases": [], "benchmark_id": "llm-stats-mmlongbench-128k", "caveat": "MMLongBench-128K evaluates multimodal long-context understanding at a 128K token context length, testing how well vision-language models reason over very long mixed text and image inputs.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MMLongBench-128K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 759, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlongbench-doc", "caveat": "MMLongBench-Doc evaluates long document understanding capabilities in vision-language models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MMLongBench-Doc", "organization_count": 1, "organizations": ["llm_stats"], "rank": 760, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlongbench-doc?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1275-mmlongbench-doc", "caveat": "MMLongBench-Doc is a long-context multi-modal benchmark comprising 1,062 expert-annotated questions. MMLONGBENCH-DOC是一个长上下文的多模态基准，由1062个专家注释的问题组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLongBench-Doc"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "MMLongBench-Doc", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 761, "released": "2024-07-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLongBench-Doc"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu", "caveat": "Massive Multitask Language Understanding benchmark testing knowledge across 57 diverse subjects including STEM, humanities, social sciences, and professional domains", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "MMLU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 762, "released": "2020-09-07", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-498-mmlu", "caveat": "MMLU (Massive Multitask Language Understanding) is a new benchmark designed to measure knowledge acquired during pretraining by evaluating models exclusively in zero-shot and few-shot settings. This makes the benchmark more challenging and more similar to how we evaluate humans. The benchmark covers 57 subjects across STEM, the humanities, the social sciences, and more. It ranges in difficulty from an elementary level to an advanced professional level, and it tests both world knowledge and problem solving ability. Subjects range from traditional areas, such as mathematics and history, to more specialized areas like law and ethics. The granularity and breadth of the subjects makes the benchmark ideal for identifying a model’s blind spots. MMLU (Massive Multitask Language Understanding) 是一个新的基准测试，旨在通过在零次学习和少次学习的环境中评估模型来测量预训练期间获得的知识。这使得基准测试更具挑战性，且更接近我们评估人类的方式。该基准测试涵盖了STEM、人文学科、社会科学等57个主题。其难度范围从小学级别到专业级别，旨在测试世界知识和解决问题的能力。测试主题范围从传统领域，如数学和历史，到更专业的领域，如法律和伦理学。题目的精细度和广度使该基准测试成为识别模型盲点的理想选择。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLU"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "MMLU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 763, "released": "2020-09-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLU"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-cot", "caveat": "Chain-of-Thought variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses chain-of-thought prompting to elicit step-by-step reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "MMLU (CoT)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 764, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-%28cot%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-chat", "caveat": "Chat-format variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses conversational prompting format for model evaluation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "MMLU Chat", "organization_count": 1, "organizations": ["llm_stats"], "rank": 765, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-chat?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-french", "caveat": "French language variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This multilingual version tests model performance in French.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "MMLU French", "organization_count": 1, "organizations": ["llm_stats"], "rank": 766, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-french?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-base", "caveat": "Base version of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. Designed to comprehensively measure the breadth and depth of a model's academic and professional understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "MMLU-Base", "organization_count": 1, "organizations": ["llm_stats"], "rank": 767, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-base?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-pro", "caveat": "A more robust and challenging multi-task language understanding benchmark that extends MMLU by expanding multiple-choice options from 4 to 10, eliminating trivial questions, and focusing on reasoning-intensive tasks. Features over 12,000 curated questions across 14 domains and causes a 16-33% accuracy drop compared to original MMLU.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "MMLU-Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 768, "released": "2024-06-03", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-pro?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1276-mmlu-pro", "caveat": "MMLU-Pro is the extension of MMLU, integrating more challenging, reasoning-focused questions and expanding the choice set from four to ten options. MMLU-Pro是MMLU的扩展版本，涵盖了更具挑战性、以推理为重点的问题，并将选择集从4个选项扩展到10个选项。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMLU-Pro"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "MMLU-Pro", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 769, "released": "2024-06-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMLU-Pro"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-prox", "caveat": "Extended version of MMLU-Pro providing additional challenging multiple-choice questions for evaluating language models across diverse academic and professional domains. Built on the foundation of the Massive Multitask Language Understanding benchmark framework.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "MMLU-ProX", "organization_count": 1, "organizations": ["llm_stats"], "rank": 770, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-prox?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-redux", "caveat": "An improved version of the MMLU benchmark featuring manually re-annotated questions to identify and correct errors in the original dataset. Provides more reliable evaluation metrics for language models by addressing dataset quality issues found in the original MMLU.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MMLU-Redux", "organization_count": 1, "organizations": ["llm_stats"], "rank": 771, "released": "2024-06-06", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-redux-2-0", "caveat": "A curated version of the MMLU benchmark featuring manually re-annotated 5,700 questions across 57 subjects to identify and correct errors in the original dataset. Addresses the 6.49% error rate found in MMLU and provides more reliable evaluation metrics for language models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MMLU-redux-2.0", "organization_count": 1, "organizations": ["llm_stats"], "rank": 772, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-redux-2.0?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmlu-stem", "caveat": "STEM-focused subset of the Massive Multitask Language Understanding benchmark, evaluating language models on science, technology, engineering, and mathematics topics including physics, chemistry, mathematics, and other technical subjects.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MMLU-STEM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 773, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmlu-stem?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmmlu", "caveat": "Multilingual Massive Multitask Language Understanding dataset released by OpenAI, featuring professionally translated MMLU test questions across 14 languages including Arabic, Bengali, German, Spanish, French, Hindi, Indonesian, Italian, Japanese, Korean, Portuguese, Swahili, Yoruba, and Chinese. Contains approximately 15,908 multiple-choice questions per language covering 57 subjects.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MMMLU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 774, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmlu?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmmu", "caveat": "MMMU (Massive Multi-discipline Multimodal Understanding) is a benchmark designed to evaluate multimodal models on college-level subject knowledge and deliberate reasoning. Contains 11.5K meticulously collected multimodal questions from college exams, quizzes, and textbooks, covering six core disciplines: Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, and Tech & Engineering across 30 subjects and 183 subfields.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMMU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 775, "released": "2023-11-27", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1248-mmmu", "caveat": "MMMU is a new benchmark designed to evaluate multimodal models on massive multi-discipline tasks demanding college-level subject knowledge and deliberate reasoning. MMMU includes 11.5K meticulously collected multimodal questions from college exams, quizzes, and textbooks. MMMU用于评估多模态大模型在复杂多学科任务中的表现，包括从大学考试和教科书中精心收集的11.5K多模态问题，涵盖六个核心学科：艺术与设计、商业、科学、健康与医学、人文与社会科学以及技术与工程。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMMU"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "MMMU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 776, "released": "2023-11-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMMU"}, {"aliases": [], "benchmark_id": "llm-stats-mmmu-val", "caveat": "Validation set of the Massive Multi-discipline Multimodal Understanding and Reasoning benchmark. Features college-level multimodal questions across 6 core disciplines (Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, Tech & Engineering) spanning 30 subjects and 183 subfields with diverse image types including charts, diagrams, maps, and tables.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMMU (val)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 777, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28val%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmmu-validation", "caveat": "Validation set of the Massive Multi-discipline Multimodal Understanding and Reasoning benchmark. Features college-level multimodal questions across 6 core disciplines (Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, Tech & Engineering) spanning 30 subjects and 183 subfields with diverse image types including charts, diagrams, maps, and tables.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMMU (validation)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 778, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-%28validation%29?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-mmmu-pro", "caveat": "Visual reasoning", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/mmmu-pro"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMMU-Pro", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 779, "released": "2024-09-05", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/mmmu-pro"}, {"aliases": [], "benchmark_id": "llm-stats-mmmu-pro", "caveat": "A more robust multi-discipline multimodal understanding benchmark that enhances MMMU through a three-step process: filtering text-only answerable questions, augmenting candidate options, and introducing vision-only input settings. Achieves significantly lower model performance (16.8-26.9%) compared to original MMMU, providing more rigorous evaluation that closely mimics real-world scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMMU-Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 780, "released": "2024-09-04", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmmu-pro-with-tools", "caveat": "MMMU-Pro variant evaluated with tool access enabled.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMMU-Pro (with tools)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 781, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmu-pro-with-tools?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmmuval", "caveat": "Validation set for MMMU (Massive Multi-discipline Multimodal Understanding and Reasoning) benchmark, designed to evaluate multimodal models on massive multi-discipline tasks demanding college-level subject knowledge and deliberate reasoning across Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, and Tech & Engineering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMMUval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 782, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmmuval?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1829-mms-vpr", "caveat": "MMS-VPR is a large-scale multimodal dataset for street-level place recognition in pedestrian areas. It contains 78,575 images and 2,512 videos from 207 locations in Chengdu, China, with rich metadata and spatial graph structure. MMS-VPR 是一个面向复杂城市步行街区的 多模态街景视觉地点识别大规模数据集，采集自成都约 70,800 平方米的开放式商业街区，覆盖 207 个地点，包含 78,575 张图像和 2,512 段视频，每条数据均带有 GPS 坐标、时间戳和文本元信息。相较以车载视角和西方城市为主的传统数据集，MMS-VPR 更贴近真实、密集、多用途的街道空间，具备丰富的视角、时段和模态多样性。此外，该数据集构建了包含 81 个节点、125 条边的空间图结构，支持结构感知的地点识别方法。数据集还定义了两个子集（Edges 和Points），支持精细化和图结构评估任务，助力多模态与地理空间理解的交叉研究。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMS-VPR"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMS-VPR", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 783, "released": "2025-05-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMS-VPR"}, {"aliases": [], "benchmark_id": "llm-stats-mmsearch", "caveat": "MMSearch evaluates multimodal models on search-based retrieval and question answering tasks that require processing both visual and textual information from search results.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMSearch", "organization_count": 1, "organizations": ["llm_stats"], "rank": 784, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1584-mmsearch", "caveat": "Logo MMSearch is a multimodal search benchmark crafted to evaluate the potential of LMMs to function as a multimodal AI search engine. This benchmark encompasses a meticulously collected dataset of 300 queries spanning 14 subfields. MMSearch 是一个多模态搜索基准，旨在评估大型语言模型（LMMs）作为多模态 AI 搜索引擎的潜力。该基准包含了一个精心收集的包含 300 个查询的数据集，涵盖 14 个子领域。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMSearch"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMSearch", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 785, "released": "2024-09-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMSearch"}, {"aliases": [], "benchmark_id": "llm-stats-mmsearch-plus", "caveat": "MMSearch-Plus is an extended variant of MMSearch with harder multimodal search and retrieval tasks requiring deeper reasoning over visual and textual search results.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMSearch-Plus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 786, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsearch-plus?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1922-mmsi-bench", "caveat": "MMSI-Bench is a novel Visual Question Answering (VQA) benchmark specifically designed to evaluate Multi-image Spatial Intelligence in multimodal large language models (MLLMs). Unlike traditional datasets that focus on spatial reasoning within a single image, MMSI-Bench emphasizes real-world inspired MMSI-Bench 是一个全新的多模态空间智能视觉问答（VQA）基准数据集，专为评估多图像空间推理能力而设计。与专注于单图像关系推理的传统数据集不同，MMSI-Bench 更贴近现实世界，聚焦于需在多张图像之间进行逻辑推理的复杂场景。该数据集由六位三维视觉专家耗时超过300小时构建，精心整理出包含1,000个高质量选择题的问题集，题目来自12万余张图像，并附有精心设计的误导选项与逐步推理过程。对34个主流开源和闭源多模态大语言模型的实证评估显示：最先进的模型准确率仅为30%至40%，而人类表现高达97%，揭示该任务的巨大挑战性与模型发展空间。此外，MMSI-Bench 配套提供自动化错误分析", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMSI-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMSI-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 787, "released": "2025-05-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMSI-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-mmsibench", "caveat": "MMSIBench is a multimodal spatial-intelligence benchmark evaluating spatial reasoning over images.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMSIBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 788, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmsibench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmstar", "caveat": "MMStar is an elite vision-indispensable multimodal benchmark comprising 1,500 challenge samples meticulously selected by humans to evaluate 6 core capabilities and 18 detailed axes. The benchmark addresses issues of visual content unnecessity and unintentional data leakage in existing multimodal evaluations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMStar", "organization_count": 1, "organizations": ["llm_stats"], "rank": 789, "released": "2024-04-09", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmstar?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1175-mmstar", "caveat": "MMStar is an elite vision-indispensable multi-modal benchmark comprising 1,500 samples meticulously selected by humans. MMStar benchmarks 6 core capabilities and 18 detailed axes, aiming to evaluate LVLMs’ multi-modal capacities with carefully balanced and purified samples. MMStar 是一个多模态基准，包含 1,500 个经过人工精心挑选的样本。MMStar 评估 6 项核心能力和 18 个详细维度，旨在通过精心平衡和净化的样本，评估 LVLM 的多模态能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMStar"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MMStar", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 790, "released": "2024-04-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMStar"}, {"aliases": [], "benchmark_id": "llm-stats-mmt-bench", "caveat": "MMT-Bench is a comprehensive multimodal benchmark for evaluating Large Vision-Language Models towards multitask AGI. It comprises 31,325 meticulously curated multi-choice visual questions from various multimodal scenarios such as vehicle driving and embodied navigation, covering 32 core meta-tasks and 162 subtasks in multimodal understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMT-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 791, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmt-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1364-mmt-bench", "caveat": "MMT-Bench is designed to assess LVLMs across massive multimodal tasks requiring expert knowledge and deliberate visual recognition, localization, reasoning, and planning. It comprises 31,325 multi-choice visual questions, covering 32 core meta-tasks and 162 subtasks in multimodal understanding. MMT-Bench考察多模态大模型的视觉识别、定位、推理和规划能力，包括31325个多选视觉问题，涵盖了32个核心元任务和162个多模态理解子任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMT-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MMT-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 792, "released": "2024-04-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMT-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1736-mmtb", "caveat": "Large language models (LLMs) demonstrate strong potential as agents for tool invocation due to their advanced comprehension and planning capabilities. Users increasingly rely on LLM-based agents to solve complex missions through iterative interactions. However, existing benchmarks predominantly acce 大型语言模型（LLM）凭借其先进的理解能力和规划能力，在作为工具调用的智能体方面展现出了巨大的潜力。用户越来越依赖基于大型语言模型的智能体，通过迭代交互来解决复杂任务。 然而，现有的基准测试主要是在单任务场景中评估智能体，无法体现现实世界的复杂性。为了填补这一空白，我们提出了 Multi-Mission Tool Bench 测试。在这个基准测试中，每个测试用例都包含多个相互关联的任务。这种设计要求智能体能够动态适应不断变化的需求。此外，所提出的基准测试探索了在固定任务数量下所有可能的任务切换模式。具体而言，我们提出了一个多智能体数据生成框架来构建这个基准测试。我们还提出了一种利用动态决策树来", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MMTB"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "MMTB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 793, "released": "2025-03-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MMTB"}, {"aliases": [], "benchmark_id": "llm-stats-mmvet", "caveat": "MM-Vet is an evaluation benchmark that examines large multimodal models on complicated multimodal tasks requiring integrated capabilities. It assesses six core vision-language capabilities: recognition, knowledge, spatial awareness, language generation, OCR, and math through questions that require one or more of these capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MMVet", "organization_count": 1, "organizations": ["llm_stats"], "rank": 794, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvet?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmvetgpt4turbo", "caveat": "MM-Vet evaluation using GPT-4 Turbo for scoring. This variant of MM-Vet examines large multimodal models on complicated multimodal tasks requiring integrated capabilities across six core vision-language abilities: recognition, knowledge, spatial awareness, language generation, OCR, and math.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MMVetGPT4Turbo", "organization_count": 1, "organizations": ["llm_stats"], "rank": 795, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvetgpt4turbo?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mmvu", "caveat": "MMVU (Multimodal Multi-disciplinary Video Understanding) is a benchmark for evaluating multimodal models on video understanding tasks across multiple disciplines, testing comprehension and reasoning capabilities on video content.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MMVU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 796, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mmvu?top_n=500"}, {"aliases": ["MMVU"], "benchmark_id": "mmvu", "caveat": "Video understanding benchmark. For non-video-native models, uses 1 fps frame extraction. Scores depend on frame sampling strategy and context length.", "document_count": 1, "document_ids": ["model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0008271298593879239, "domain": "vision", "name": "MMVU", "organization_count": 1, "organizations": ["Z.ai"], "rank": 797, "released": "2025-06-01", "source": "model_reports", "url": "https://github.com/MMVU/MMVU"}, {"aliases": [], "benchmark_id": "llm-stats-mobileminiwob-sr", "caveat": "MobileMiniWob++ SR (Success Rate) is an adaptation of the MiniWob++ web interaction benchmark for mobile Android environments within AndroidWorld. It comprises 92 web interaction tasks adapted for touch-based mobile interfaces, evaluating agents' ability to navigate and interact with web applications on mobile devices.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MobileMiniWob++_SR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 798, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileminiwob%2B%2B-sr?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mobileworld", "caveat": "MobileWorld is a benchmark for evaluating multimodal agents on real mobile-device tasks, testing GUI grounding, navigation, and multi-step task completion in mobile environments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MobileWorld", "organization_count": 1, "organizations": ["llm_stats"], "rank": 799, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mobileworld?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1273-molpuzzle", "caveat": "MolPuzzle is a benchmark for evaluating MLM's reasoning ability, comprising 234 instances of structure elucidation, which feature over 18,000 QA samples. MolPuzzle用于考察MLM的推理能力，包含234个结构解析实例以及超过18000个QA样本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MolPuzzle"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MolPuzzle", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 800, "released": "2024-09-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MolPuzzle"}, {"aliases": [], "benchmark_id": "opencompass-1704-mono2stereo", "caveat": "For evaluating stereo image conversion, it provides test data for five different scenarios: animation, indoor, outdoor, complex, and simple, with a total of approximately 2,500 test sample pairs. It also offers evaluation metrics for assessing the stereo effect. 用于测评立体影像转换，提供了动画，室内，室外，复杂，简单共五种场景的测试数据，总共约2500对测试样本。并提供了用于测评立体效果的评价指标-立体交并比。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Mono2Stereo"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "Mono2Stereo", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 801, "released": "2025-03-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Mono2Stereo"}, {"aliases": [], "benchmark_id": "opencompass-1999-morse-500", "caveat": "MORSE-500 introduces 500 programmatically generated videos testing 6 reasoning types: abstract, physical, planning, spatial, temporal, mathematical. Its controllable generation (via Manim, Matplotlib, generative models etc.) enables scalable difficulty, designed to evolve as SOTA models improve. 当前多模态推理基准存在三大不足：依赖静态图像、偏重数学解题、易饱和。为此，我们提出MORSE-500视频基准：包含500个脚本化视频片段，涵盖抽象、物理、规划、空间、时间、数学六类推理问题。其核心在于程序化生成（使用Manim、Matplotlib等），可精确控制视觉复杂度、干扰物密度和时序动态，从而系统化提升难度。与易过时的静态基准不同，MORSE-500具备可持续演进能力，其可控生成流程能无限创建新挑战实例。在顶尖模型（Gemini 2.5 Pro、OpenAI o3等）上的测试揭示了显著性能差距，尤其在抽象和规划任务上。我们开源数据集、生成脚本及评估工具，以促进透明、可复现的前沿研究。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MORSE-500"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "MORSE-500", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 802, "released": "2025-06-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MORSE-500"}, {"aliases": [], "benchmark_id": "llm-stats-motionbench", "caveat": "MotionBench is a benchmark for evaluating multimodal models on motion understanding in videos, testing the ability to comprehend temporal dynamics, movement patterns, and action sequences.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MotionBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 803, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/motionbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1682-motionbench", "caveat": "MotionBench, is a comprehensive evaluation benchmark designed to assess the fine-grained motion comprehension of video understanding models. MotionBench是一个综合性的评估基准，旨在评估视频理解模型的细粒度运动理解能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MotionBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MotionBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 804, "released": "2025-01-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MotionBench"}, {"aliases": [], "benchmark_id": "opencompass-2154-motionmillion", "caveat": "MotionMillion: The largest open-sourced 3D human motion dataset with text annotation, including 2.5k hours 1.9M episodes. MotionMillion：目前最大的开源带文本标注的 3D 人体动作数据集，包含 2500 小时、190 万个片段。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MotionMillion"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MotionMillion", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 805, "released": "2025-07-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MotionMillion"}, {"aliases": [], "benchmark_id": "opencompass-930-mr-ben-meta-reasoning-benchmark", "caveat": "Welcome to the dataset page for the Meta-Reasoning Benchmark associated with our recent publication `Mr-Ben: A Comprehensive Meta-Reasoning Benchmark for Large Language Models`. We have provided a demo evaluate script for you to try out benchmark in mere two steps. We encourage everyone to try out our benchmark in the SOTA models and return its results to us. We would be happy to include it in the eval_results and update the evaluation tables below for you. 本工作联合MIT,清华,剑桥等知名院校, 提出了一个评测大语言模型对复杂问题的推理过程的“阅卷”批改能力的评测数据集，有别于以前的以结果匹配为评测模式的数据集MR-Ben，我们的数据集基于GSM8K[1], MMLU[2], LogiQA[3], MHPP[4]等数据集经由细致的高水平人工标注构建而成，显著地增加了难度及区分度。我们细致地分析了包括claude3.5, GPT4-Turbo, Kimi, Zhipu, Yi-Large, Qwen2, DeepseekCoderv2 等国内外一线的大语言模型，发现开源的模型在复杂推理的场景下有望追上顶尖的闭源模型。该评测数据集的所有数据均已开源，并且支持一键评测。欢迎所有做大模型训练的小伙伴向我们分享你的评测结果，我们会及时更新榜单。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MR-Ben-Meta-Reasoning-Benchmark"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MR-Ben-Meta-Reasoning-Benchmark", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 806, "released": "2024-06-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MR-Ben-Meta-Reasoning-Benchmark"}, {"aliases": [], "benchmark_id": "opencompass-1587-mr-gsm8k", "caveat": "MR-GSM8K is a challenging benchmark designed to evaluate the meta-reasoning capabilities of state-of-the-art Large Language Models (LLMs). It goes beyond traditional evaluation metrics by focusing on the reasoning process rather than just the final answer. MR-GSM8K 是一个旨在评估最先进大型语言模型（LLMs）元推理能力的挑战性基准。它超越了传统的评估指标，专注于推理过程而非仅仅关注最终答案，从而对模型的认知能力进行更细致的评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MR-GSM8K"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MR-GSM8K", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 807, "released": "2023-12-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MR-GSM8K"}, {"aliases": [], "benchmark_id": "opencompass-1586-mrag-bench", "caveat": "MRAG-Bench consists of 16,130 images and 1,353 human-annotated multiple-choice questions across 9 distinct scenarios, providing a robust and systematic evaluation of Large Vision Language Model (LVLM)’s vision-centric multimodal retrieval-augmented generation (RAG) abilities. MRAG-Bench 包含 16,130 张图片和 1,353 个跨越 9 个不同场景的人标注多选题，为大型视觉语言模型（LVLM）的视觉中心多模态检索增强生成（RAG）能力提供了稳健和系统的评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MRAG-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MRAG-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 808, "released": "2024-10-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MRAG-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr", "caveat": "MRCR (Multi-Round Coreference Resolution) is a synthetic long-context reasoning task where models must navigate long conversations to reproduce specific model outputs. It tests the ability to distinguish between similar requests and reason about ordering while maintaining attention across extended contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 809, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-128k-2-needle", "caveat": "MRCR (Multi-Round Coreference Resolution) at 128K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 2 items to retrieve.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 128K (2-needle)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 810, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%282-needle%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-128k-4-needle", "caveat": "MRCR (Multi-Round Coreference Resolution) at 128K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 4 items to retrieve.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 128K (4-needle)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 811, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%284-needle%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-128k-8-needle", "caveat": "MRCR (Multi-Round Coreference Resolution) at 128K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 8 items to retrieve.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 128K (8-needle)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 812, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-128k-%288-needle%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-1m", "caveat": "MRCR 1M is a variant of the Multi-Round Coreference Resolution benchmark designed for testing extremely long context capabilities with approximately 1 million tokens. It evaluates models' ability to maintain reasoning and attention across ultra-long conversations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 1M", "organization_count": 1, "organizations": ["llm_stats"], "rank": 813, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-1m-pointwise", "caveat": "MRCR 1M (pointwise) is a variant of the Multi-Round Coreference Resolution benchmark that uses pointwise evaluation for ultra-long contexts (~1M tokens). This version evaluates each response independently rather than comparatively, testing models' absolute performance on long-context reasoning tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 1M (pointwise)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 814, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-1m-%28pointwise%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-64k-2-needle", "caveat": "MRCR (Multi-Round Coreference Resolution) at 64K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 2 items to retrieve.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 64K (2-needle)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 815, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%282-needle%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-64k-4-needle", "caveat": "MRCR (Multi-Round Coreference Resolution) at 64K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 4 items to retrieve.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 64K (4-needle)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 816, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%284-needle%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-64k-8-needle", "caveat": "MRCR (Multi-Round Coreference Resolution) at 64K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 8 items to retrieve.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR 64K (8-needle)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 817, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-64k-%288-needle%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-v2", "caveat": "MRCR v2 (Multi-Round Coreference Resolution version 2) is an enhanced version of the synthetic long-context reasoning task. It extends the original MRCR framework with improved evaluation criteria and additional complexity for testing models' ability to maintain attention and reasoning across extended contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR v2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 818, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-v2-8-needle", "caveat": "MRCR v2 (8-needle) is a variant of the Multi-Round Coreference Resolution benchmark that includes 8 needle items to retrieve from long contexts. This tests models' ability to simultaneously track and reason about multiple pieces of information across extended conversations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR v2 (8-needle)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 819, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-%288-needle%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mrcr-v2-8-needle-512k-1m", "caveat": "MRCR v2 8-needle variant evaluated on contexts from 512K to 1M tokens.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MRCR v2 (8-needle, 512K-1M)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 820, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mrcr-v2-8-needle-512k-1m?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1102-ms-marco", "caveat": "MS MARCO comprises of 1,010,916 anonymized questions—sampled from Bing’s search query logs—each with a human generated answer and 182,669 completely human rewritten generated answers. In addition, the dataset contains 8,841,823 passages. MS MARCO 数据集包含 1,010,916 个来自 Bing 的搜索查询日志的匿名问题，每个问题都有一个人工生成的答案和 182,669 个完全由人重写的生成答案。此外，该数据集还包含从 3,563,535 个由 Bing 检索的网页文档中提取的 8,841,823 个段落，这些段落提供了策划自然语言答案所需的信息。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MS_MARCO"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "MS_MARCO", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 821, "released": "2018-10-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MS_MARCO"}, {"aliases": [], "benchmark_id": "llm-stats-msqa", "caveat": "MSQA is a multilingual question-answering benchmark that measures knowledge and reasoning across a diverse set of languages.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MSQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 822, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/msqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mt-aime-2025", "caveat": "MT-AIME 2025 is Cohere's internal multilingual translation of AIME 2025, evaluated for Arabic, Japanese, and Korean in the Command A+ release.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "MT-AIME 2025", "organization_count": 1, "organizations": ["llm_stats"], "rank": 823, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-aime-2025?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mt-bench", "caveat": "MT-Bench is a challenging multi-turn benchmark that measures the ability of large language models to engage in coherent, informative, and engaging conversations. It uses strong LLMs as judges for scalable and explainable evaluation of multi-turn dialogue capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MT-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 824, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mt-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1172-mt-bench-101", "caveat": "MT-Bench-101 is specifically designed to evaluate the finegrained abilities of LLMs in multi-turn dialogues. MT-Bench-101 专门设计用于评估 LLMs 在多轮对话中的细粒度能力。通过对真实多轮对话数据的详细分析，构建了一个三层级的能力分类法，涵盖 1388 个多轮对话中的 4208 个轮次，涉及 13 种不同的任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MT-Bench-101"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "MT-Bench-101", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 825, "released": "2024-06-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MT-Bench-101"}, {"aliases": [], "benchmark_id": "opencompass-1929-mtcmb", "caveat": "We introduce MTCMB-a Multi-Task Benchmark for Evaluating LLMs on TCM Knowledge, Reasoning, and Safety. Developed in collaboration with certified TCM experts, MTCMB comprises 12 sub-datasets spanning five major categories: knowledge QA, language understanding, diagnostic reasoning, formula generation 大语言模型在中医领域的应用日益增多，到底大语言模型在这一古老的学科表现如何？虽然有一些零碎的评测数据集，但大都以单选题、病案分析、粗糙的医患对话为主，很难全面评估大语言模型在中医领域的实际能力。近日，中山大学联合湖南中医药大学等团队推出全球首个中医多任务评测基准（Benchmark） —— MTCMB: A Multi-Task Benchmark Framework for Evaluating LLMs on Knowledge, Reasoning, and Safety in Traditional Chinese Medicine", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MTCMB"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "MTCMB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 826, "released": "2025-05-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MTCMB"}, {"aliases": [], "benchmark_id": "opencompass-2034-mteb", "caveat": "MTEB is a benchmark for evaluating text embedding models, aiming to enhance long-term usability and reproducibility, focusing on task extensibility, data integrity validation, and result generalizability. MTEB 是一个评估文本嵌入模型的基准，旨在提升其长期可用性和可复现性，涵盖任务扩展性、数据完整性验证和结果通用性等维度。该基准包含包括分类、检索和聚类在内的多种任务，覆盖多个语言和领域。MTEB 引入了自动化测试管道、持续集成框架和社区贡献机制，以支持基准的可扩展性和质量控制。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MTEB"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "MTEB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 827, "released": "2025-06-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MTEB"}, {"aliases": ["MTOB", "Machine Translation from One Book"], "benchmark_id": "mtob", "caveat": "Translation from a single grammar book, reported as half-book and full-book variants that are not interchangeable.", "document_count": 1, "document_ids": ["model_reports:meta_llama_4"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "MTOB", "organization_count": 1, "organizations": ["Meta"], "rank": 828, "released": "2023-09-28", "source": "model_reports", "url": "https://arxiv.org/abs/2309.16575"}, {"aliases": [], "benchmark_id": "llm-stats-mtvqa", "caveat": "MTVQA (Multilingual Text-Centric Visual Question Answering) is the first benchmark featuring high-quality human expert annotations across 9 diverse languages, consisting of 6,778 question-answer pairs across 2,116 images. It addresses visual-textual misalignment problems in multilingual text-centric VQA.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MTVQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 829, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mtvqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1502-mtvqa", "caveat": "MTVQA is designed to evaluate LMMs' multilingual text understanding, featuring high-quality human expert annotations across 9 diverse languages. MTVQA用于评估多模态大模型理解多语言文本的能力，包含来自9种语言的由人类专家注释的高质量数据。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MTVQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MTVQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 830, "released": "2024-06-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MTVQA"}, {"aliases": [], "benchmark_id": "llm-stats-muirbench", "caveat": "A comprehensive benchmark for robust multi-image understanding capabilities of multimodal LLMs. Consists of 12 diverse multi-image tasks involving 10 categories of multi-image relations (e.g., multiview, temporal relations, narrative, complementary). Comprises 11,264 images and 2,600 multiple-choice questions created in a pairwise manner, where each standard instance is paired with an unanswerable variant for reliable assessment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MuirBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 831, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/muirbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-multichallenge", "caveat": "MultiChallenge is a realistic multi-turn conversation evaluation benchmark that challenges frontier LLMs across four key categories: instruction retention (maintaining instructions throughout conversations), inference memory (recalling and connecting details from previous turns), reliable versioned editing (adapting to evolving instructions during collaborative editing), and self-coherence (avoiding contradictions in responses). The benchmark evaluates models on sustained, contextually complex dialogues across diverse topics including travel planning, technical documentation, and professional communication.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Multi-Challenge", "organization_count": 1, "organizations": ["llm_stats"], "rank": 832, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multichallenge?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-multi-if", "caveat": "Multi-IF benchmarks LLMs on multi-turn and multilingual instruction following. It expands upon IFEval by incorporating multi-turn sequences and translating English prompts into 7 other languages, resulting in 4,501 multilingual conversations with three turns each. The benchmark reveals that current leading LLMs struggle with maintaining accuracy in multi-turn instructions and shows higher error rates for non-Latin script languages.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Multi-IF", "organization_count": 1, "organizations": ["llm_stats"], "rank": 833, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-if?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-multi-swe-bench", "caveat": "A multilingual benchmark for issue resolving that evaluates Large Language Models' ability to resolve software issues across diverse programming ecosystems. Covers 7 programming languages (Java, TypeScript, JavaScript, Go, Rust, C, and C++) with 1,632 high-quality instances carefully annotated by 68 expert annotators. Addresses limitations of existing benchmarks that focus almost exclusively on Python.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Multi-SWE-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 834, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multi-swe-bench?top_n=500"}, {"aliases": ["MultiChallenge"], "benchmark_id": "multichallenge", "caveat": "Multi-turn instruction retention, graded by an LLM judge.", "document_count": 1, "document_ids": ["model_reports:qwen3_5_model_card"], "document_share": 0.0008271298593879239, "domain": "instruction_following", "name": "MultiChallenge", "organization_count": 1, "organizations": ["Qwen"], "rank": 835, "released": "2025-01-29", "source": "model_reports", "url": "https://arxiv.org/abs/2501.17399"}, {"aliases": [], "benchmark_id": "llm-stats-multilf", "caveat": "MultiLF benchmark", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "MultiLF", "organization_count": 1, "organizations": ["llm_stats"], "rank": 836, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multilf?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-multilingual-mgsm-cot", "caveat": "Multilingual Grade School Math (MGSM) benchmark evaluates language models' chain-of-thought reasoning abilities across ten typologically diverse languages. Contains 250 grade-school math problems manually translated from GSM8K dataset into languages including Bengali and Swahili.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "Multilingual MGSM (CoT)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 837, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mgsm-%28cot%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-multilingual-mmlu", "caveat": "MMLU-ProX is a comprehensive multilingual benchmark covering 29 typologically diverse languages, building upon MMLU-Pro. Each language version consists of 11,829 identical questions enabling direct cross-linguistic comparisons. The benchmark evaluates large language models' reasoning capabilities across linguistic and cultural boundaries through challenging, reasoning-focused questions with 10 answer choices.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Multilingual MMLU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 838, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multilingual-mmlu?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1751-multiloko", "caveat": "MultiLoKo is a multilingual knowledge benchmark, covering 30 languages plus English. MultiLoKo是一个多语言知识基准，涵盖30种语言及英语。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MultiLoKo"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "MultiLoKo", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 839, "released": "2025-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MultiLoKo"}, {"aliases": [], "benchmark_id": "llm-stats-multipl-e", "caveat": "MultiPL-E is a scalable and extensible system for translating unit test-driven code generation benchmarks to multiple programming languages. It extends HumanEval and MBPP Python benchmarks to 18 additional programming languages, enabling evaluation of neural code generation models across diverse programming paradigms and language features.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "MultiPL-E", "organization_count": 1, "organizations": ["llm_stats"], "rank": 840, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-multipl-e-humaneval", "caveat": "MultiPL-E is a scalable and extensible approach to benchmarking neural code generation that translates unit test-driven code generation benchmarks across multiple programming languages. It extends the HumanEval benchmark to 18 additional programming languages, enabling evaluation of code generation models across diverse programming paradigms and providing insights into how models generalize programming knowledge across language boundaries.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "Multipl-E HumanEval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 841, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-humaneval?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-multipl-e-mbpp", "caveat": "MultiPL-E extends the Mostly Basic Python Problems (MBPP) benchmark to 18+ programming languages for evaluating multilingual code generation capabilities. MBPP contains 974 crowd-sourced programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality. Each problem includes a task description, code solution, and automated test cases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Multipl-E MBPP", "organization_count": 1, "organizations": ["llm_stats"], "rank": 842, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/multipl-e-mbpp?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-musiccaps", "caveat": "MusicCaps is a dataset composed of 5,521 music examples, each labeled with an English aspect list and a free text caption written by musicians. The dataset contains 10-second music clips from AudioSet paired with rich textual descriptions that capture sonic qualities and musical elements like genre, mood, tempo, instrumentation, and rhythm. Created to support research in music-text understanding and generation tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MusicCaps", "organization_count": 1, "organizations": ["llm_stats"], "rank": 843, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/musiccaps?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-musr", "caveat": "MuSR (Multistep Soft Reasoning) is a benchmark for evaluating language models on multistep soft reasoning tasks specified in natural language narratives. Created through a neurosymbolic synthetic-to-natural generation algorithm, it generates complex reasoning scenarios like murder mysteries roughly 1000 words in length that challenge current LLMs including GPT-4. The benchmark tests chain-of-thought reasoning capabilities across domains involving commonsense reasoning about physical and social situations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "MuSR", "organization_count": 1, "organizations": ["llm_stats"], "rank": 844, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/musr?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-mvbench", "caveat": "A comprehensive multi-modal video understanding benchmark covering 20 challenging video tasks that require temporal understanding beyond single-frame analysis. Tasks span from perception to cognition, including action recognition, temporal reasoning, spatial reasoning, object interaction, scene transition, and counterfactual inference. Uses a novel static-to-dynamic method to systematically generate video tasks from existing annotations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "MVBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 845, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/mvbench?top_n=500"}, {"aliases": ["MVbench", "MVBench"], "benchmark_id": "mvbench", "caveat": "Video understanding benchmark. For non-video-native models, uses 1 fps frame extraction. Scores depend on frame sampling strategy and context length.", "document_count": 1, "document_ids": ["model_reports:zai_glm_5_3_flash_model_card"], "document_share": 0.0008271298593879239, "domain": "vision", "name": "MVbench", "organization_count": 1, "organizations": ["Z.ai"], "rank": 846, "released": "2025-06-01", "source": "model_reports", "url": "https://github.com/MVBench/MVBench"}, {"aliases": [], "benchmark_id": "opencompass-1509-mvbench", "caveat": "MVBench can test MLLMs' temporal understanding in the dynamic video tasks. It covers 20 challenging video tasks that cannot be effectively solved with a single frame. MVBench用于评估多模态大模型在动态视频任务中的时间理解能力，由20个单帧内容无法有效解决的挑战性的视频任务组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MVBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MVBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 847, "released": "2023-11-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MVBench"}, {"aliases": [], "benchmark_id": "opencompass-1537-mvl-sib", "caveat": "MVL-SIB is a multilingual dataset that provides image-sentence pairs spanning 205 languages and 7 topical categories (entertainment, geography, health, politics, science, sports, travel). It was constructed by extending the SIB-200 benchmark. MVL-SIB 是一个多语言数据集，提供了涵盖 205 种语言和 7 个主题类别的图像-句子对（ entertainment ， geography ， health ， politics ， science ， sports ， travel ）。它通过扩展 SIB-200 基准构建而成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MVL-SIB"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MVL-SIB", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 848, "released": "2025-02-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MVL-SIB"}, {"aliases": [], "benchmark_id": "opencompass-1905-mvpbench", "caveat": "MVPBench, a curated benchmark designed to rigorously evaluate visual physical reasoning through the lens of visual CoT. Each example features interleaved multi-image inputs and demands not only the correct final answer but also a coherent, step-by-step reasoning path grounded in evolving visual cues MVPBench专注于视觉物理推理中的视觉链式思维（CoT）能力评估。该基准涵盖真实图像、多步逻辑与多条可行思维路径，每个样例均配有图像证据，要求模型在剥离文本提示依赖的前提下，不仅得出正确答案，还需正确完成每一个中间推理步骤，模拟人类的逐步图像推理过程。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/MVPBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "MVPBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 849, "released": "2025-06-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/MVPBench"}, {"aliases": [], "benchmark_id": "llm-stats-nanogpt", "caveat": "NanoGPT is an OpenAI AI-self-improvement evaluation that measures whether models can optimize training recipes for small GPT-style models, part of the suite tracking progress toward accelerating internal research.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "NanoGPT", "organization_count": 1, "organizations": ["llm_stats"], "rank": 850, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nanogpt?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-natural-questions", "caveat": "Natural Questions is a question answering dataset featuring real anonymized queries issued to Google search engine. It contains 307,373 training examples where annotators provide long answers (passages) and short answers (entities) from Wikipedia pages, or mark them as unanswerable.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Natural Questions", "organization_count": 1, "organizations": ["llm_stats"], "rank": 851, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/natural-questions?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-natural2code", "caveat": "NaturalCodeBench (NCB) is a challenging code benchmark designed to mirror the complexity and variety of real-world coding tasks. It comprises 402 high-quality problems in Python and Java, selected from natural user queries from online coding services, covering 6 different domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Natural2Code", "organization_count": 1, "organizations": ["llm_stats"], "rank": 852, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/natural2code?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1081-naturalcodebench", "caveat": "NaturalCodeBench is challenging code benchmark designed to mirror the complexity and variety of scenarios in real coding tasks. NCB comprises 402 high-quality problems in Python and Java, meticulously selected from natural user queries from online coding services, covering 6 different domains. NaturalCodeBench 是一个具有挑战性的代码基准，旨在反映真实编码任务中的复杂性和多样性。NaturalCodeBench 包含 402 个高质量的 Python 和 Java 问题，这些问题是从在线编码服务的自然用户查询中精心挑选的，涵盖了 6 个不同的领域。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NaturalCodeBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "NaturalCodeBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 853, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/NaturalCodeBench"}, {"aliases": [], "benchmark_id": "opencompass-1117-naturalproofs", "caveat": "NATURALPROOFS is a multi-domain corpus of mathematical statements and their proofs, written in natural mathematical language. NATURALPROOFS unifies broad coverage, deep coverage, and low-resource mathematical sources, allowing for evaluating both in-distribution and zero-shot generalization. NATURALPROOFS 是一个多领域的数学语句及其证明的语料库，采用自然数学语言编写。整合了广泛覆盖、深入覆盖和低资源数学来源，便于评估分布内和零样本泛化的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NaturalProofs"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "NaturalProofs", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 854, "released": "2021-06-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/NaturalProofs"}, {"aliases": [], "benchmark_id": "opencompass-1074-newsbench", "caveat": "NewsBench is a novel evaluation framework to systematically assess the capabilities of Large Language Models (LLMs) for editorial capabilities in Chinese journalism. NewsBench 是一个新颖的评估框架，旨在系统性地评估大型语言模型在中文新闻编辑能力上的表现。构建的基准数据集聚焦于写作能力的四个方面和安全遵循的六个方面，包含 1,267 个手动精心设计的测试样本，类型包括选择题和简答题，涵盖 24 个新闻领域的五项编辑任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NewsBench"], "document_share": 0.0008271298593879239, "domain": "创作", "name": "NewsBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 855, "released": "2024-06-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/NewsBench"}, {"aliases": [], "benchmark_id": "llm-stats-nexus", "caveat": "NexusRaven benchmark for evaluating function calling capabilities of large language models in zero-shot scenarios across cybersecurity tools and API interactions", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "Nexus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 856, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nexus?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-nih-multi-needle", "caveat": "Multi-needle in a haystack benchmark for evaluating long-context comprehension capabilities of language models by testing retrieval of multiple target pieces of information from extended documents", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "NIH/Multi-needle", "organization_count": 1, "organizations": ["llm_stats"], "rank": 857, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nih-multi-needle?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-nl2repo", "caveat": "NL2Repo evaluates long-horizon coding capabilities including repository-level understanding, where models must generate or modify code across entire repositories from natural language specifications.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "NL2Repo", "organization_count": 1, "organizations": ["llm_stats"], "rank": 858, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nl2repo?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-nmos", "caveat": "NMOS evaluation benchmark for assessing model performance on specialized tasks", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "NMOS", "organization_count": 1, "organizations": ["llm_stats"], "rank": 859, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nmos?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-community-2256e9c9-b256-4444-b639-7cc3b1855d96", "caveat": null, "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A2256e9c9-b256-4444-b639-7cc3b1855d96?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "nolima", "organization_count": 1, "organizations": ["llm_stats"], "rank": 860, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A2256e9c9-b256-4444-b639-7cc3b1855d96?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-nolima-128k", "caveat": "NoLiMa evaluated at a 131072-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "NoLiMa 128K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 861, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-nolima-32k", "caveat": "NoLiMa evaluated at a 32768-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "NoLiMa 32K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 862, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-32k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-nolima-64k", "caveat": "NoLiMa evaluated at a 65536-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "NoLiMa 64K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 863, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nolima-64k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-nova-63", "caveat": "NOVA-63 is a multilingual evaluation benchmark covering 63 languages, designed to assess LLM performance across diverse linguistic contexts and tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500"], "document_share": 0.0008271298593879239, "domain": "general", "name": "NOVA-63", "organization_count": 1, "organizations": ["llm_stats"], "rank": 864, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nova-63?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1783-nppc", "caveat": "Nondeterministic Polynomial-time Problem Challenge (NPPC), an ever-scaling reasoning benchmark for LLMs. 非确定性多项式时间问题挑战 （NPPC），这是一个不断扩展的 LLM 推理基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NPPC"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "NPPC", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 865, "released": "2025-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/NPPC"}, {"aliases": [], "benchmark_id": "llm-stats-nq", "caveat": "Natural Questions (NQ) benchmark containing real user questions issued to Google search with answers found from Wikipedia, designed for training and evaluation of automatic question answering systems", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "NQ", "organization_count": 1, "organizations": ["llm_stats"], "rank": 866, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nq?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-513-nq", "caveat": "NQ (NaturalQuestion) corpus contains questions from real users, and it requires QA systems to read and comprehend an entire Wikipedia article that may or may not contain the answer to the question. The inclusion of real user questions, and the requirement that solutions should read an entire page to find the answer, cause NQ to be a more realistic and challenging task than prior QA datasets. NQ 数据集来自于真实用户的问题，它要求 QA 系统阅读和理解整个维基百科文章，这些文章可能包含也可能不包含问题的答案。由真实用户问题构成，以及需要阅读整个页面才能找到答案的要求，比以往的 QA 数据集更现实和更具挑战性的任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NQ"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "NQ", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 867, "released": "2019-06-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/NQ"}, {"aliases": [], "benchmark_id": "llm-stats-nuscene", "caveat": "A multimodal benchmark for scene understanding and reasoning over the nuScenes autonomous driving domain.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Nuscene", "organization_count": 1, "organizations": ["llm_stats"], "rank": 868, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/nuscene?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1533-nutritionqa", "caveat": "CoSyn-400K dataset contains 9 categories of synthetc text-rich images with 2.7M instruction-tuning data 通过代码引导的合成多模态数据生成扩展文本丰富图像理解，CoSyn-400K 数据集包含 9 类合成文本丰富图像，以及 270 万条指令微调数据。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/NutritionQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "NutritionQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 869, "released": "2025-02-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/NutritionQA"}, {"aliases": [], "benchmark_id": "llm-stats-objectron", "caveat": "Objectron evaluates 3D object detection and pose estimation capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500"], "document_share": 0.0008271298593879239, "domain": "3d", "name": "Objectron", "organization_count": 1, "organizations": ["llm_stats"], "rank": 870, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/objectron?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-525-ocnli", "caveat": "OCNLI is a Chinese natural language inference task, which requires to determine the logical relation between two sentences, with three relations: entailment, contradiction and neutral. OCNLI是一个中文自然语言推理任务，要求根据两个句子判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OCNLI"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "OCNLI", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 871, "released": "2020-10-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OCNLI"}, {"aliases": [], "benchmark_id": "llm-stats-ocrbench", "caveat": "OCRBench: Comprehensive evaluation benchmark for assessing Optical Character Recognition (OCR) capabilities in Large Multimodal Models across text recognition, scene text VQA, and document understanding tasks", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "image_to_text", "name": "OCRBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 872, "released": "2024-01-17", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-557-ocrbench", "caveat": "OCRBench provides a comprehensive evaluation of Large Multimodal Models, such as GPT4V and Gemini, in various text-related visual tasks including Text Recognition, Scene Text-Centric Visual Question Answering (VQA), Document-Oriented VQA, Key Information Extraction (KIE), and Handwritten Mathematical Expression Recognition (HMER). OCRBench对 GPT4V 和 Gemini 等大型多模态模型在各种文本相关的视觉任务中的表现进行了全面的评估，包括文本识别、场景文本为中心的视觉问答 (VQA)、面向文档的 VQA、关键信息提取 (KIE) 和手写数学表达式识别 (HMER)。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OCRBench"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "OCRBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 873, "released": "2024-01-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OCRBench"}, {"aliases": [], "benchmark_id": "llm-stats-ocrbench-v2-en", "caveat": "OCRBench v2 English subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with English text content", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "image_to_text", "name": "OCRBench-V2 (en)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 874, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28en%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ocrbench-v2-zh", "caveat": "OCRBench v2 Chinese subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with Chinese text content", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "image_to_text", "name": "OCRBench-V2 (zh)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 875, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2-%28zh%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ocrbench-v2", "caveat": "OCRBench v2: Enhanced large-scale bilingual benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with 10,000 human-verified question-answering pairs across 8 core OCR capabilities", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "image_to_text", "name": "OCRBench_V2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 876, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ocrbench-v2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-octocodingbench", "caveat": "Octopus coding benchmark for evaluating multi-language programming capabilities", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "OctoCodingBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 877, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/octocodingbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-odinw", "caveat": "Object Detection in the Wild (ODinW) benchmark for evaluating object detection models' task-level transfer ability across diverse real-world datasets in terms of prediction accuracy and adaptation efficiency", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500"], "document_share": 0.0008271298593879239, "domain": "vision", "name": "ODinW", "organization_count": 1, "organizations": ["llm_stats"], "rank": 878, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/odinw?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-officeqa-pro", "caveat": "OfficeQA Pro evaluates AI models on professional knowledge-work questions and tasks drawn from real office workflows, including document analysis, spreadsheet reasoning, and information synthesis across business domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "OfficeQA Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 879, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/officeqa-pro?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ojbench", "caveat": "OJBench is a competition-level code benchmark designed to assess the competitive-level code reasoning abilities of large language models. It comprises 232 programming competition problems from NOI and ICPC, categorized into Easy, Medium, and Hard difficulty levels. The benchmark evaluates models' ability to solve complex competitive programming challenges using Python and C++.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "OJBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 880, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ojbench-cpp", "caveat": "OJBench (C++) is the C++ subset of OJBench, a competition-level code benchmark that evaluates large language models on programming competition problems from NOI and ICPC using C++ as the implementation language.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "OJBench (C++)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 881, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ojbench-cpp?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1705-olymmath", "caveat": "A benchmark of 200 Olympiad math problems across algebra, geometry, number theory, and combinatorics. Available in English and Chinese, it features two difficulty levels: EASY (AIME-level) to test standard reasoning, and HARD to challenge advanced models. 一个包含 200 道奥林匹克数学题的基准测试，涵盖代数、几何、数论和组合。我们提供英文和中文版本，并提供两个难度等级：EASY（AIME水平）用于测试标准推理能力，以及 HARD 用于挑战高级模型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OlymMATH"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "OlymMATH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 882, "released": "2025-03-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OlymMATH"}, {"aliases": [], "benchmark_id": "llm-stats-olympiadbench", "caveat": "A challenging benchmark for promoting AGI with Olympiad-level bilingual multimodal scientific problems. Comprises 8,476 math and physics problems from international and Chinese Olympiads and the Chinese college entrance exam, featuring expert-level annotations for step-by-step reasoning. Includes both text-only and multimodal problems in English and Chinese.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "OlympiadBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 883, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/olympiadbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1070-olympiadbench", "caveat": "OlympiadBench, an Olympiad-level bilingual multimodal scientific benchmark, featuring\n8,476 problems from Olympiad-level mathematics and physics competitions, including the\nChinese college entrance exam. Each problem is detailed with expert-level annotations\nfor step-by-step reasoning. OlympiadBench 是一个奥林匹克级别的双语多模态科学基准，包含来自奥林匹克级数学和物理竞赛的8,476道题目，包括中国高考。每道题目都配有专家级别的注释，提供逐步推理的详细说明。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OlympiadBench"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "OlympiadBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 884, "released": "2024-06-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OlympiadBench"}, {"aliases": [], "benchmark_id": "opencompass-1329-olympicarena", "caveat": "OlympicArena evaluates cognitive reasoning abilities. It includes 11,163 bilingual problems spanning seven fields and 62 international Olympic competitions. OlympicArena用于评估大模型的认知推理能力，包含来自7个领域、62项国际奥林匹克比赛的11163个双语问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OlympicArena"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "OlympicArena", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 885, "released": "2024-06-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OlympicArena"}, {"aliases": [], "benchmark_id": "opencompass-1244-omni-math", "caveat": "Omni-MATH focuses exclusively on mathematics and comprises a vast collection of 4428 competition-level problems with rigorous human annotation. These problems are meticulously categorized into over 33 sub-domains and span more than 10 distinct difficulty levels Omni-MATH用于评估LLM在奥林匹克水平上的数学推理能力，包括4428道竞赛级问题，并带有严格的人工注释。这些问题被精心分类为超过33个子领域，涵盖10多个不同的难度级别。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Omni-MATH"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "Omni-MATH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 886, "released": "2024-10-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Omni-MATH"}, {"aliases": [], "benchmark_id": "opencompass-1553-omnialign-v", "caveat": "OmniAlign-V datasets mainly focus on improving the alignment of Multi-modal Large Language Models(MLLMs) with human preference. It contains 205k high-quality Image-Quetion-Answer pairs with open-ended, creative quetions and long, knowledge-rich, comprehensive answers. OmniAlign-V 数据集主要关注提高多模态大型语言模型（MLLMs）与人类偏好的对齐。它包含 205k 个高质量的图像-问答对，包含开放式、创意性问题以及长篇、知识丰富、内容全面的答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniAlign-V"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "OmniAlign-V", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 887, "released": "2025-02-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniAlign-V"}, {"aliases": [], "benchmark_id": "llm-stats-omnibench", "caveat": "A novel multimodal benchmark designed to evaluate large language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. Comprises 1,142 question-answer pairs covering 8 task categories from basic perception to complex inference, with a unique constraint that accurate responses require integrated understanding of all three modalities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OmniBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 888, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1981-omnibench", "caveat": "OmniBench is a multi-dimensional benchmark for virtual agents, designed to systematically evaluate ten core capabilities such as planning, decision-making, and instruction comprehension through automatically generated task graphs with controllable complexity. OmniBench 是一个面向虚拟智能体的多维度评测基准，旨在通过自动化流程生成具有可控复杂度的任务图，系统评估智能体在计划、决策、指令理解等十个核心能力上的表现。该基准包含 36,000 个图结构任务，覆盖 20 个真实场景，并引入 OmniEval 框架，实现子任务级别的细粒度评估，显著提升了评测的效率和可扩展性。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "OmniBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 889, "released": "2025-06-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniBench"}, {"aliases": [], "benchmark_id": "llm-stats-omnibench-music", "caveat": "Music component of OmniBench, a comprehensive benchmark for evaluating omni-language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. The music category includes various compositions and performances that require integrated understanding across text, image, and audio modalities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OmniBench Music", "organization_count": 1, "organizations": ["llm_stats"], "rank": 890, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnibench-music?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-omnidocbench", "caveat": "OmniDocBench evaluates multimodal models on document understanding tasks such as OCR, layout parsing, and structured document comprehension.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OmniDocBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 891, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1942-omnidocbench", "caveat": "OmniDocBench is a comprehensive benchmark for evaluating document parsing in real-world scenarios. It includes 981 PDF pages across 9 document types, annotated with dense paragraph-level bboxes with text and attributes. Along with its designed evaluation methods, it provides Fine-grained results. OmniDocBench是一个用于评估真实场景下多样性文档解析效果的评测集，它包含了981个页面，覆盖9种文档类型（包括研报、教材、报纸、手写笔记、杂志等），具有段落级别的位置标注和内容标注，还有阅读顺序标注和属性标注，并开发了配套的评测方法，使其既具备单模块的评测能力（包括布局检测，公式识别，表格识别，文本识别），又具备端到端的评测能力，针对不同元素提供了分页面以及分属性的精细化评测结果，精准定位模型文档解析的痛点问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniDocBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "OmniDocBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 892, "released": "2024-12-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniDocBench"}, {"aliases": [], "benchmark_id": "llm-stats-omnidocbench-1-5", "caveat": "OmniDocBench 1.5 is a comprehensive benchmark for evaluating multimodal large language models on document understanding tasks, including OCR, document parsing, information extraction, and visual question answering across diverse document types. Lower Overall Edit Distance scores are better.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OmniDocBench 1.5", "organization_count": 1, "organizations": ["llm_stats"], "rank": 893, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnidocbench-1.5?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-omnigaia", "caveat": "OmniGAIA evaluates multimodal perception and reasoning in agentic contexts, testing a model's ability to process diverse inputs and perform complex multi-step reasoning tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OmniGAIA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 894, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnigaia?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1801-omnigirl", "caveat": "A multi-modal, multi-language benchmark dataset for the GitHub Issue Resolution task. 一个面向 GitHub Issue ResoLution任务的多语言、多模态基准数据集，包含以下特点: 1.支持 Python、Java、JS、TS 四种主流编程语言，2. 输入信息涵盖文本、图像、网页等多种模态，3. 提供可复现的 Docker 评估环境。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniGIRL"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "OmniGIRL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 895, "released": "2025-05-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniGIRL"}, {"aliases": [], "benchmark_id": "llm-stats-omnimath", "caveat": "A Universal Olympiad Level Mathematic Benchmark for Large Language Models containing 4,428 competition-level problems with rigorous human annotation, categorized into over 33 sub-domains and spanning more than 10 distinct difficulty levels", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "OmniMath", "organization_count": 1, "organizations": ["llm_stats"], "rank": 896, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omnimath?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2142-omnimmi", "caveat": "OmniMMI is a cutting-edge benchmark for multi-modal interaction, specifically designed for OmniLLMs in streaming video environments. It includes 1,121 videos and 2,290 questions, tackling the challenges of streaming video understanding and proactive reasoning across six unique subtasks. OmniMMI is a cutting-edge benchmark for multi-modal interaction, specifically designed for OmniLLMs in streaming video environments. It includes 1,121 videos and 2,290 questions, tackling the challenges of streaming video understanding and proactive reasoning across six unique subtasks.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OmniMMI"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "OmniMMI", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 897, "released": "2025-04-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OmniMMI"}, {"aliases": [], "benchmark_id": "llm-stats-omniscience", "caveat": "OmniScience is a broad scientific knowledge and reasoning benchmark that measures both answer accuracy and non-hallucination (calibrated abstention) across science domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "OmniScience", "organization_count": 1, "organizations": ["llm_stats"], "rank": 898, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-omniscience-non-hallucination-rate", "caveat": "OmniScience variant that reports the non-hallucination rate, defined as one minus the hallucination rate.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "OmniScience (non-hallucination rate)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 899, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/omniscience-non-hallucination-rate?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-onemillion-bench", "caveat": "OneMillion Bench evaluates AI agents on high-economic-value tasks that require sustained, reliable execution across long-horizon real-world workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "OneMillion Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 900, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/onemillion-bench?top_n=500"}, {"aliases": ["OneMillionBench", "One Million Bench"], "benchmark_id": "onemillionbench", "caveat": "Long-context recall and use; a \"with tools\" figure is not comparable to a no-tools run.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "OneMillionBench", "organization_count": 1, "organizations": ["Tencent"], "rank": 901, "released": null, "source": "model_reports", "url": "https://github.com/humanlaya/OneMillion-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-open-rewrite", "caveat": "OpenRewriteEval is a benchmark for evaluating open-ended rewriting of long-form texts, covering a wide variety of rewriting types expressed through natural language instructions including formality, expansion, conciseness, paraphrasing, and tone and style transfer.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "Open-rewrite", "organization_count": 1, "organizations": ["llm_stats"], "rank": 902, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/open-rewrite?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-openai-mmlu", "caveat": "MMLU (Massive Multitask Language Understanding) is a comprehensive benchmark that measures a text model's multitask accuracy across 57 diverse academic and professional subjects. The test covers elementary mathematics, US history, computer science, law, morality, business ethics, clinical knowledge, and many other domains spanning STEM, humanities, social sciences, and professional fields. To attain high accuracy, models must possess extensive world knowledge and problem-solving ability.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "OpenAI MMLU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 903, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mmlu?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-openai-mrcr-2-needle-128k", "caveat": "Multi-round Co-reference Resolution (MRCR) benchmark for evaluating an LLM's ability to distinguish between multiple needles hidden in long context. Models are given a long, multi-turn synthetic conversation and must retrieve a specific instance of a repeated request, requiring reasoning and disambiguation skills beyond simple retrieval.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "OpenAI-MRCR: 2 needle 128k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 904, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-openai-mrcr-2-needle-1m", "caveat": "Multi-Round Co-reference Resolution benchmark that tests an LLM's ability to distinguish between multiple similar needles hidden in long conversations. Models must reproduce specific instances of content (e.g., 'Return the 2nd poem about tapirs') from multi-turn synthetic conversations, requiring reasoning about context, ordering, and subtle differences between similar outputs.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "OpenAI-MRCR: 2 needle 1M", "organization_count": 1, "organizations": ["llm_stats"], "rank": 905, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-1m?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-openai-mrcr-2-needle-256k", "caveat": "Multi-Round Co-reference Resolution (MRCR) benchmark that tests long-context reasoning by evaluating a model's ability to distinguish between similar outputs, reason about ordering, and reproduce specific content from multi-turn conversations containing multiple writing requests on overlapping topics at 256k tokens.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "OpenAI-MRCR: 2 needle 256k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 906, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-mrcr%3A-2-needle-256k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-openbookqa", "caveat": "OpenBookQA is a question-answering dataset modeled after open book exams for assessing human understanding. It contains 5,957 multiple-choice elementary-level science questions that probe understanding of 1,326 core science facts and their application to novel situations, requiring combination of open book facts with broad common knowledge through multi-hop reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "OpenBookQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 907, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openbookqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-518-openbookqa", "caveat": "OpenBookQA contains questions that require multi-step reasoning, application of common-sense knowledge, and in-depth comprehension of text. It is a new type of question-answering dataset, modeled after open-book exams, designed to assess human understanding of a specific topic. OpenBookQA包含需要多步推理、运用常识知识、深入理解文本等能力的问题，是一种新型的问答数据集，其模式借鉴了开放式书本考试，用于评估人类对某一主题理解的程度。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenbookQA"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "OpenbookQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 908, "released": "2018-09-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenbookQA"}, {"aliases": [], "benchmark_id": "opencompass-631-openfindata", "caveat": "OpenFinData is an open-source financial evaluation dataset jointly released by EastMoney and Shanghai Artificial Intelligence Laboratory. This dataset represents the most realistic industrial scenario requirements and is currently the most comprehensive and professional financial evaluation dataset. OpenFinData是由东方财富与上海人工智能实验室联合发布的开源金融评测数据集。该数据集代表了最真实的产业场景需求，是目前场景最全、专业性最深的金融评测数据集。它基于东方财富实际金融业务的多样化丰富场景，旨在为金融科技领域的研究者和开发者提供一个高质量的数据资源。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenFinData"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "OpenFinData", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 909, "released": "2023-12-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenFinData"}, {"aliases": [], "benchmark_id": "llm-stats-openrca", "caveat": "OpenRCA is a benchmark for evaluating AI models on root cause analysis tasks. For each failure case, the model receives 1 point if all generated root-cause elements match the ground-truth ones, and 0 points if any mismatch is identified. The overall accuracy is the average score across all failure cases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "OpenRCA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 910, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openrca?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1754-openturingbench", "caveat": "OpenTuringBench, a new benchmark based on OLLMs, designed to train and evaluate machine-generated text detectors on the Turing Test and Authorship Attribution problems. OpenTuringBench，这是一个基于OLLMs的新基准测试，旨在训练和评估机器生成文本检测器在图灵测试和作者归属问题上的性能。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenTuringBench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "OpenTuringBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 911, "released": "2025-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenTuringBench"}, {"aliases": [], "benchmark_id": "opencompass-1979-openunlearning", "caveat": "OpenUnlearning is an efficient and modular benchmark platform designed to support and drive research on \"unlearning\" in large language models (LLMs). OpenUnlearning 是一个高效且模块化的基准平台，旨在支持和推动大型语言模型（LLM）中的“遗忘”（unlearning）研究。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OpenUnlearning"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "OpenUnlearning", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 912, "released": "2025-06-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OpenUnlearning"}, {"aliases": [], "benchmark_id": "opencompass-1990-opt-bench", "caveat": "OPT-BENCH is a comprehensive benchmark designed to evaluate large language model (LLM) agents on large-scale search space optimization problems, focusing on their iterative reasoning and problem-solving capabilities. OPT-BENCH 是一个面向大型语言模型（LLM）智能体的大规模搜索空间优化评测基准，旨在系统评估模型在迭代推理和解决复杂优化问题中的能力。 该基准包含 30 个任务，包括 20 个来自 Kaggle 的真实机器学习任务和 10 个经典 NP 问题，涵盖预测建模、图论和组合优化等领域。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OPT-BENCH"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "OPT-BENCH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 913, "released": "2025-06-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OPT-BENCH"}, {"aliases": [], "benchmark_id": "llm-stats-community-64d67847-06bd-423a-923c-c2acfab82281", "caveat": null, "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3A64d67847-06bd-423a-923c-c2acfab82281?top_n=500"], "document_share": 0.0008271298593879239, "domain": "other", "name": "OptimBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 914, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3A64d67847-06bd-423a-923c-c2acfab82281?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2086-or-bench", "caveat": "OR Bench is a large-scale benchmark designed to evaluate the excessive rejection behavior of large language models. OR-Bench 是一个旨在评估大型语言模型过度拒绝行为的大规模基准。它衡量LLM在过度拒绝和有害提示拒绝方面的表现，涵盖暴力、隐私等10个类别。基准包含8万个、1千个困难及6百个有害提示，通过Mixtral等工具自动化生成。它为未来安全对齐研究提供强大测试平台。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OR-Bench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "OR-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 915, "released": "2024-05-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OR-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1917-orak", "caveat": "Orak (오락) is a foundational benchmark for evaluating Large Language Model (LLM) agents in diverse popular video games. Orak是一个基础性的基准,用于评估在各种流行视频游戏中的大型语言模型(LLM)代理。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Orak"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "Orak", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 916, "released": "2025-06-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Orak"}, {"aliases": [], "benchmark_id": "opencompass-2005-oss-bench", "caveat": "OSS-Bench replaces functions with LLM-generated code and evaluates them using three natural metrics: compilability, functional correctness, and memory safety, leveraging robust signals like compilation failures, test-suite violations, and sanitizer alerts as ground truth. OSS-Bench，这是一个基准生成器，它可以从真实的开源软件中自动构建大规模的实时评估任务。OSS-Bench 将函数替换为 LLM 生成的代码，并使用三个自然指标（可编译性、功能正确性和内存安全性）对其进行评估，并利用编译失败、测试套件违规和Sanitizer警报等稳健信号作为基准事实。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/OSS-Bench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "OSS-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 917, "released": "2025-06-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/OSS-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-osworld", "caveat": "OSWorld: The first-of-its-kind scalable, real computer environment for multimodal agents, supporting task setup, execution-based evaluation, and interactive learning across Ubuntu, Windows, and macOS with 369 computer tasks involving real web and desktop applications, OS file I/O, and multi-application workflows", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OSWorld", "organization_count": 1, "organizations": ["llm_stats"], "rank": 918, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-osworld-2-0", "caveat": "OSWorld 2.0 is a benchmark of 108 long-horizon, real-world computer-use workflows spanning everyday and professional tasks. Each task is an end-to-end workflow that takes human users a median of about 1.6 hours, scored with a binary-completion metric, and targets challenges such as dynamic environments, cross-source reasoning, and implicit-state inference.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OSWorld 2.0", "organization_count": 1, "organizations": ["llm_stats"], "rank": 919, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-2.0?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-osworld-extended", "caveat": "OSWorld is a scalable, real computer environment benchmark for evaluating multimodal agents on open-ended tasks across Ubuntu, Windows, and macOS. It comprises 369 computer tasks involving real web and desktop applications, OS file I/O, and multi-application workflows. The benchmark evaluates agents' ability to interact with computer interfaces using screenshots and actions in realistic computing environments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OSWorld Extended", "organization_count": 1, "organizations": ["llm_stats"], "rank": 920, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-extended?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-osworld-screenshot-only", "caveat": "OSWorld Screenshot-only: A variant of the OSWorld benchmark that evaluates multimodal AI agents using only screenshot observations to complete open-ended computer tasks across real operating systems (Ubuntu, Windows, macOS). Tests agents' ability to perform complex workflows involving web apps, desktop applications, file I/O, and multi-application tasks through visual interface understanding and GUI grounding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OSWorld Screenshot-only", "organization_count": 1, "organizations": ["llm_stats"], "rank": 921, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-screenshot-only?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-osworld-g", "caveat": "OSWorld-G (Grounding) evaluates screenshot grounding accuracy for OS automation tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OSWorld-G", "organization_count": 1, "organizations": ["llm_stats"], "rank": 922, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-g?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-osworld-verified", "caveat": "OSWorld-Verified is a verified subset of OSWorld, a scalable real computer environment for multimodal agents supporting task setup, execution-based evaluation, and interactive learning across Ubuntu, Windows, and macOS.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OSWorld-Verified", "organization_count": 1, "organizations": ["llm_stats"], "rank": 923, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/osworld-verified?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ovbench", "caveat": "OVBench is an online video understanding benchmark that evaluates a model's ability to perceive, memorize, and reason about real-time video streams as they unfold.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OVBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 924, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ovbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ovobench", "caveat": "OVOBench (Online Video Online Benchmark) evaluates streaming video understanding, testing a model's ability to perceive and respond to video content in real time as it unfolds.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "OVOBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 925, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ovobench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1283-p-mmeval", "caveat": "P-MMEval is a comprehensive multilingual multitask benchmark, covering effective fundamental and capability-specialized datasets. P-MMEval是一个全面的多语言多任务基准，涵盖了高效的基础和专项能力数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/P-MMEval"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "P-MMEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 926, "released": "2024-11-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/P-MMEval"}, {"aliases": [], "benchmark_id": "llm-stats-paperbench", "caveat": "PaperBench is a benchmark for evaluating AI agents on their ability to replicate research papers. It tests models on complex, multi-step workflows involving code implementation, experimentation, and reproducing scientific results from academic publications.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "PaperBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 927, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/paperbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1730-paperbench", "caveat": "PaperBench is a benchmark evaluating the ability of AI agents to replicate state-of-the-art AI research. PaperBench，这是一个评估AI代理复制最新AI研究能力的基准测试。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PaperBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "PaperBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 928, "released": "2025-04-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PaperBench"}, {"aliases": [], "benchmark_id": "opencompass-1836-pashtoocr", "caveat": "PsOCR is a large-scale synthetic dataset for Optical Character Recognition in low-resource Pashto language. PsOCR is a large-scale synthetic dataset for Optical Character Recognition in low-resource Pashto language.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PashtoOCR"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "PashtoOCR", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 929, "released": "2025-05-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PashtoOCR"}, {"aliases": [], "benchmark_id": "llm-stats-pathmcqa", "caveat": "PathMMU is a massive multimodal expert-level benchmark for understanding and reasoning in pathology, containing 33,428 multimodal multi-choice questions and 24,067 images validated by seven pathologists. It evaluates Large Multimodal Models (LMMs) performance on pathology tasks, with the top-performing model GPT-4V achieving only 49.8% zero-shot performance compared to 71.8% for human pathologists.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "PathMCQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 930, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/pathmcqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1076-pca-bench", "caveat": "PCA-Bench is a multimodal decisionmaking benchmark for evaluating the integrated capabilities of Multimodal Large Language Models (MLLMs). Departing from previous\nbenchmarks focusing on simplistic tasks and individual model capability. PCA-Bench 是一个多模态决策基准，用于评估多模态大型语言模型（MLLMs）的综合能力。与之前专注于简单任务和单个模型能力的基准不同，PCA-Bench 引入了三个复杂场景：自动驾驶、家庭机器人和开放世界游戏。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PCA-Bench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "PCA-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 931, "released": "2024-02-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PCA-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-perceptionbench", "caveat": "PerceptionBench is Moonshot AI's internal benchmark for evaluating atomic visual perception capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "PerceptionBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 932, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptionbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-perceptiontest", "caveat": "A novel multimodal video benchmark designed to evaluate perception and reasoning skills of pre-trained models across video, audio, and text modalities. Contains 11.6k real-world videos (average 23 seconds) filmed by participants worldwide, densely annotated with six types of labels. Focuses on skills (Memory, Abstraction, Physics, Semantics) and reasoning types (descriptive, explanatory, predictive, counterfactual). Shows significant performance gap between human baseline (91.4%) and state-of-the-art video QA models (46.2%).", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "PerceptionTest", "organization_count": 1, "organizations": ["llm_stats"], "rank": 933, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/perceptiontest?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1959-personalens", "caveat": "PersonaLens  a large-scale benchmark specifically designed to evaluate personalization in task-oriented dialogues. The benchmark features 1,500 in-depth user profiles, each integrating real demographic data, detailed cross-domain preferences, and rich interaction histories. PersonaLens是一个专为任务导向型对话设计的、大规模的个性化能力评测基准。它包含1,500个深度用户画像，每个画像都集成了真实的人口统计信息、详尽的个人偏好及历史互动记录。这些画像与覆盖20个领域的111项真实世界任务相结合，并辅以动态的“情景上下文”来模拟现实世界的复杂性。为了实现自动化、可扩展的评估，该基准引入了两个LLM驱动的智能体：一个“用户智能体”负责模拟真人与AI进行对话，另一个“评判智能体”则对个性化水平、任务成功率和对话质量进行系统性打分。PersonaLens旨在为研究社区提供一个强大可靠的工具，共同推动下一代更懂你、更智能的AI助手的研发。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PersonaLens"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "PersonaLens", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 934, "released": "2025-06-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PersonaLens"}, {"aliases": [], "benchmark_id": "opencompass-2087-perteval-scfm", "caveat": "PertEval-scFM is a standardized framework to evaluate single-cell foundation models (scFMs) for predicting cellular perturbation effects. PertEval-scFM 是一个评估单细胞基础模型（scFM）扰动效应预测的标准化框架。它从分布偏移泛化性、扰动强度及上下文对齐三方面评测。测试集包含Norman、Replogle等数据集，涵盖数千扰动样本和基因。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PertEval-scFM"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "PertEval-scFM", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 935, "released": "2024-10-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PertEval-scFM"}, {"aliases": [], "benchmark_id": "llm-stats-phibench", "caveat": "PhiBench is an internal benchmark designed to evaluate diverse skills and reasoning abilities of language models, covering a wide range of tasks including coding (debugging, extending incomplete code, explaining code snippets) and mathematics (identifying proof errors, generating related problems). Created by Microsoft's research team to address limitations of standard academic benchmarks and guide the development of the Phi-4 model.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "PhiBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 936, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/phibench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-phybench", "caveat": "PHYBench is a benchmark of real-world physics problems spanning mechanics, electromagnetism, thermodynamics, optics, and modern physics, designed to evaluate physical perception and multi-step quantitative reasoning in large language models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "physics", "name": "PHYBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 937, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/phybench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2080-phygenbench", "caveat": "PhyGenBench is a benchmark designed to evaluate physical commonsense correctness in text-to-video generation models. I PhyGenBench 是一个评估T2V模型物理常识的基准。它涵盖力学、光学、热学和材料特性四大领域27种物理定律，包含160个提示。结合PhyGenEval框架，利用视听和大语言模型自动化评估模型的物理理解能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PhyGenBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "PhyGenBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 938, "released": "2024-10-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PhyGenBench"}, {"aliases": [], "benchmark_id": "llm-stats-physicsfinals", "caveat": "PHYSICS is a comprehensive benchmark for university-level physics problem solving, containing 1,297 expert-annotated problems covering six core areas: classical mechanics, quantum mechanics, thermodynamics and statistical mechanics, electromagnetism, atomic physics, and optics. Each problem requires advanced physics knowledge and mathematical reasoning. Even advanced models like o3-mini achieve only 59.9% accuracy.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "PhysicsFinals", "organization_count": 1, "organizations": ["llm_stats"], "rank": 939, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/physicsfinals?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1576-physreason", "caveat": "PhysReason is a comprehensive physics-based reasoning benchmark consisting of 1,200 physics problems spanning multiple domains, with a focus on both knowledge-based (25%) and reasoning-based (75%) questions. PhysReason 是一个包含 1,200 个物理问题的综合物理推理基准，涵盖多个领域，重点关注基于知识（25%）和推理（75%）的问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PhysReason"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "PhysReason", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 940, "released": "2025-02-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PhysReason"}, {"aliases": [], "benchmark_id": "opencompass-2350-picabench", "caveat": null, "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PICABench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "PICABench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 941, "released": "2025-10-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PICABench"}, {"aliases": [], "benchmark_id": "llm-stats-pinchbench", "caveat": "PinchBench evaluates coding agents on real-world agentic coding tasks, measuring both best-case and average performance across complex software engineering scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "PinchBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 942, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/pinchbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-piqa", "caveat": "PIQA (Physical Interaction: Question Answering) is a benchmark dataset for physical commonsense reasoning in natural language. It tests AI systems' ability to answer questions requiring physical world knowledge through multiple choice questions with everyday situations, focusing on atypical solutions inspired by instructables.com. The dataset contains 21,000 multiple choice questions where models must choose the most appropriate solution for physical interactions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "physics", "name": "PIQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 943, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/piqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-532-piqa", "caveat": "PIQA is a physical interaction question answering task, which requires to select the most reasonable solution based on the given scenario and two possible solutions. This task is designed to test the model's knowledge in physical commonsense. This dataset consists of 16k training samples, 800 development samples and 2k test samples, all on English text. PIQA是一个物理交互问答任务，要求根据给定的场景和两个可能的解决方案，选择最合理的方案。这个任务是为了测试模型在物理常识方面的知识。这个数据集包含了16000个训练样本，800个开发样本和2000个测试样本，所有的文本都是英文文本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PIQA"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "PIQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 944, "released": "2019-11-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PIQA"}, {"aliases": [], "benchmark_id": "opencompass-1251-planbench", "caveat": "PlanBench is an extensible benchmark suite based on the kinds of domains used in the automated planning community, especially in the International Planning Competition, to test the capabilities of LLMs in planning or reasoning about actions and change. PlanBench用于评估LLM的规划能力，基于自动化规划社区（尤其是在国际规划竞赛）中涉及的各种领域来测试大模型在规划或推理行动和变更方面的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PlanBench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "PlanBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 945, "released": "2022-06-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PlanBench"}, {"aliases": [], "benchmark_id": "llm-stats-plawbench", "caveat": "PLawBench evaluates language models on professional legal knowledge and reasoning tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "PLawBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 946, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/plawbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-pmc-vqa", "caveat": "A medical visual question answering benchmark built on biomedical literature and medical figures.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "PMC-VQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 947, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/pmc-vqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-pointgrounding", "caveat": "PointArena is a comprehensive platform for evaluating multimodal pointing across diverse reasoning scenarios. It includes Point-Bench, a curated dataset of ~1,000 pointing tasks across five categories: Spatial (positional references), Affordance (functional part identification), Counting (attribute-based grouping), Steerable (relative pointing), and Reasoning (open-ended visual inference). The benchmark evaluates language-guided pointing capabilities in vision-language models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "PointGrounding", "organization_count": 1, "organizations": ["llm_stats"], "rank": 948, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/pointgrounding?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1675-pokerbench", "caveat": "PokerBench contains natural language game scenarios and optimal decisions computed by solvers in No Limit Texas Hold’em. It is divided into pre-flop and post-flop datasets, each with training and test splits. PokerBench包含自然语言游戏场景和由求解器在无限制德州扑克中计算出的最优决策。它分为前注和后注数据集，每个数据集都包含训练集和测试集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PokerBench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "PokerBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 949, "released": "2025-01-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PokerBench"}, {"aliases": [], "benchmark_id": "llm-stats-polymath", "caveat": "Polymath is a challenging multi-modal mathematical reasoning benchmark designed to evaluate the general cognitive reasoning abilities of Multi-modal Large Language Models (MLLMs). The benchmark comprises 5,000 manually collected high-quality images of cognitive textual and visual challenges across 10 distinct categories, including pattern recognition, spatial reasoning, and relative reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "PolyMATH", "organization_count": 1, "organizations": ["llm_stats"], "rank": 950, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-polymath-en", "caveat": "PolyMath is a multilingual mathematical reasoning benchmark covering 18 languages and 4 difficulty levels from easy to hard, ensuring difficulty comprehensiveness, language diversity, and high-quality translation. The benchmark evaluates mathematical reasoning capabilities of large language models across diverse linguistic contexts, making it a highly discriminative multilingual mathematical benchmark.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "PolyMath-en", "organization_count": 1, "organizations": ["llm_stats"], "rank": 951, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/polymath-en?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-pope", "caveat": "Polling-based Object Probing Evaluation (POPE) is a benchmark for evaluating object hallucination in Large Vision-Language Models (LVLMs). POPE addresses the problem where LVLMs generate objects inconsistent with target images by using a polling-based query method that asks yes/no questions about object presence in images, providing more stable and flexible evaluation of object hallucination.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "POPE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 952, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/pope?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1362-pope", "caveat": "POPE is an improved evaluation method for LVLMs' object hallucination by proposing a polling-based query method. It offers a more stable and flexible solution. POPE用于评估视觉语言模型的物体幻觉，基于轮询的查询方法设计，提供了一种更稳定、更灵活的评估方案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/POPE"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "POPE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 953, "released": "2023-05-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/POPE"}, {"aliases": [], "benchmark_id": "llm-stats-popqa", "caveat": "PopQA is an entity-centric open-domain question-answering dataset consisting of 14,000 QA pairs designed to evaluate language models' ability to memorize and recall factual knowledge across entities with varying popularity levels. The dataset probes both parametric memory (stored in model parameters) and non-parametric memory effectiveness, with questions covering 16 diverse relationship types from Wikidata converted to natural language using templates. Created by sampling knowledge triples from Wikidata and converting them to natural language questions, focusing on long-tail entities to understand LMs' strengths and limitations in memorizing factual knowledge.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "PopQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 954, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/popqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1579-postersum", "caveat": "The PosterSum dataset is a multimodal benchmark designed for the summarization of scientific posters into research paper abstracts. The dataset consists of 16,305 research posters collected from major machine learning conferences, including ICLR, ICML, and NeurIPS, spanning the years 2022-2024. PosterSum 数据集是一个多模态基准数据集，旨在将科学海报总结成研究论文摘要。该数据集包含从2022-2024年的主要机器学习会议（包括 ICLR、ICML 和 NeurIPS）收集的 16,305 篇研究海报。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PosterSum"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "PosterSum", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 955, "released": "2025-02-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PosterSum"}, {"aliases": [], "benchmark_id": "llm-stats-posttrainbench", "caveat": "PostTrainBench evaluates a model's ability to autonomously post-train base models. Given pretrain-only base models, the agent must complete the full pipeline of data synthesis, training, evaluation, and iteration within a time budget, scored across downstream benchmarks such as AIME2025, BFCL, GPQA Main, GSM8K, and HumanEval.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "PostTrainBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 956, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-posttrainbench-lite", "caveat": "PostTrainBench Lite measures whether an agent can design and execute a full post-training strategy (data, prompts, RL recipe, and eval loop) for a pretrained base model under a constrained time budget, scored as normalized mean reward over the improvement window.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "PostTrainBench Lite", "organization_count": 1, "organizations": ["llm_stats"], "rank": 957, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/posttrainbench-lite?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-prbench-finance", "caveat": "PRBench-Finance evaluates professional reasoning on finance tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "PRBench-Finance", "organization_count": 1, "organizations": ["llm_stats"], "rank": 958, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-finance?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-prbench-legal", "caveat": "PRBench-Legal evaluates professional reasoning on legal tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "PRBench-Legal", "organization_count": 1, "organizations": ["llm_stats"], "rank": 959, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/prbench-legal?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-presentbench", "caveat": "PresentBench evaluates AI agents on producing presentation-style deliverables, such as generating lesson-plan slides and structured documents from source materials.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "PresentBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 960, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/presentbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1681-prmbench-preview", "caveat": "PRMBench is a benchmark dataset for evaluating process-level reward models (PRMs). It consists of 6,216 data instances, each containing a question, a solution process, and a modified process with errors. PRMBench 是一个用于评估过程级奖励模型（PRM）的基准数据集。它包含 6,216 个数据实例，每个实例包含一个问题、一个解决方案过程以及一个包含错误的修改过程。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/PRMBench_Preview"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "PRMBench_Preview", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 961, "released": "2025-01-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/PRMBench_Preview"}, {"aliases": [], "benchmark_id": "opencompass-1632-probench", "caveat": "ProBench is a benchmark that contains open-ended multimodal queries that require intensive expert-level knowledge to solve. ProBench是一个包含需要大量专家级知识来解决的开放式多模态查询的基准。ProBench 包含 10 个任务领域和 56 个子领域，支持 17 种语言，并支持最多 13 轮对话。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ProBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 962, "released": "2025-03-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ProBench"}, {"aliases": [], "benchmark_id": "opencompass-1621-processbench", "caveat": "ProcessBench can measure the ability to identify erroneous steps in mathematical reasoning. It consists of 3,400 test cases, primarily focused on competition- and Olympiad-level math problems. ProcessBench，用于衡量识别数学推理中错误步骤的能力。它包含 3,400 个测试案例，主要关注竞赛和奥林匹克级别的数学问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProcessBench"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "ProcessBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 963, "released": "2024-12-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ProcessBench"}, {"aliases": [], "benchmark_id": "llm-stats-profbench", "caveat": "ProfBench evaluates models on professional-domain reasoning and knowledge-work tasks, including search-augmented question answering across expert fields.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ProfBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 964, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/profbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-program-bench", "caveat": "Program Bench evaluates code-generation agents by asking them to recreate a program's behavior from only a compiled binary and documentation. It spans 200 tasks from small CLI tools to large systems such as FFmpeg and SQLite, with submissions judged against more than 248,000 fuzz-generated behavioral tests.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Program Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 965, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/program-bench?top_n=500"}, {"aliases": [], "benchmark_id": "programbench", "caveat": "Cleanroom program-rebuild tasks; most models score low, so small differences sit within run-to-run noise.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "coding", "name": "ProgramBench", "organization_count": 1, "organizations": ["Tencent"], "rank": 966, "released": "2026-05-05", "source": "model_reports", "url": "https://arxiv.org/abs/2605.03546"}, {"aliases": [], "benchmark_id": "opencompass-1633-projudge", "caveat": "ProJudge is a comprehensive, multi-modal, multi-discipline, and multi-difficulty benchmark specifically designed for evaluating abilities of MLLM-based process judges.It comprises 2,400 test cases and 50,118 step-level labels, spanning four scientific disciplines with diverse difficulty levels and ProJudge 是一个针对基于 MLLM 的过程裁判能力的全面、多模态、多学科和多难度的基准。它包含 2,400 个测试案例和 50,118 个步骤级标签，涵盖四个科学学科，难度级别和内容多样化。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProJudge"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ProJudge", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 967, "released": "2025-03-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ProJudge"}, {"aliases": [], "benchmark_id": "opencompass-1120-proofnet", "caveat": "ProofNet is a benchmark for autoformalization and formal proving of undergraduate-level mathematics. The ProofNet benchmarks consists of 371 examples, each consisting of a formal theorem statement in Lean 3, a natural language theorem statement, and a natural language proof. ProofNet 是一个用于本科数学的自动形式化和形式证明的基准。包含 371 个示例，每个示例包括一个 Lean 3 中的形式定理陈述、一个自然语言定理陈述和一个自然语言证明。这些问题主要来自流行的本科纯数学教材，涵盖实分析、复分析、线性代数、抽象代数和拓扑等主题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ProofNet"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "ProofNet", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 968, "released": "2023-02-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ProofNet"}, {"aliases": [], "benchmark_id": "llm-stats-protocolqa", "caveat": "ProtocolQA is a multiple-choice benchmark on troubleshooting failed experimental outcomes from common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "ProtocolQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 969, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/protocolqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1000-q-bench", "caveat": "Q-Bench/Q-Bench+ is a benchmark for general-purpose foundation models on low-level vision. Q-Bench/Q-Bench+是一个面向多模态大模型底层视觉理解的数据集。此数据集从底层视觉的感知、描述、评价能力出发来对多模态大模型进行完整的测试，测试的对象既包括单张图像也包括图像对。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Q-Bench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "Q-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 970, "released": "2024-08-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Q-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1103-qasc", "caveat": "QASC is a question-answering dataset with a focus on sentence composition. It consists of 9,980 8-way multiple-choice questions about grade school science (8,134 train, 926 dev, 920 test), and comes with a corpus of 17M sentences. QASC 是一个专注于句子组合的问答数据集。它包含 9,980 道小学科学的多项选择题（8,134 道用于训练，926 道用于开发，920 道用于测试），并配有一个包含 1,700 万个句子的语料库。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/QASC"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "QASC", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 971, "released": "2020-02-04", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/QASC"}, {"aliases": [], "benchmark_id": "llm-stats-qasper", "caveat": "QASPER is a dataset of 5,049 information-seeking questions and answers anchored in 1,585 NLP research papers. Questions are written by NLP practitioners who read only titles and abstracts, while answers require understanding the full paper text and provide supporting evidence. The dataset challenges models with complex reasoning across document sections for academic document question answering. Each question seeks information present in the full text and is answered by a separate set of NLP practitioners who also provide supporting evidence to answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "Qasper", "organization_count": 1, "organizations": ["llm_stats"], "rank": 972, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qasper?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qmsum", "caveat": "QMSum is a benchmark for query-based multi-domain meeting summarization consisting of 1,808 query-summary pairs over 232 meetings across academic, product, and committee domains. The dataset enables models to select and summarize relevant spans of meetings in response to specific queries. Published at NAACL 2021, QMSum presents significant challenges in long meeting summarization where models must identify and summarize relevant content based on user queries.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "QMSum", "organization_count": 1, "organizations": ["llm_stats"], "rank": 973, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qmsum?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qvhighlights", "caveat": "QVHighlights is a video moment retrieval benchmark for detecting moments and highlights in videos via natural language queries. Given a query, the model must localize the start and end times of relevant moments in the video, evaluated using metrics such as Recall@1 at a 0.5 IoU threshold.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "QVHighlights", "organization_count": 1, "organizations": ["llm_stats"], "rank": 974, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qvhighlights?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qwenclawbench", "caveat": "QwenClawBench is a real-user-distribution Claw agent benchmark for evaluating coding agents on realistic developer tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "QwenClawBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 975, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenclawbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qwen-qoder-bench", "caveat": "QwenQoderBench is Qwen's internal benchmark for evaluating coding-agent performance.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "QwenQoderBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 976, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-qoder-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qwen-react-bench", "caveat": "QwenReactBench is Qwen's internal React application generation benchmark, reported as a BT/Elo rating.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "QwenReactBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 977, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-react-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qwen-svg", "caveat": "QwenSVG is Qwen's internal SVG generation benchmark for evaluating front-end and visual code generation. Scores are reported as BT/Elo ratings from auto-rendered outputs judged by a multimodal evaluator.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "QwenSVG", "organization_count": 1, "organizations": ["llm_stats"], "rank": 978, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-svg?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qwen-swe-bench", "caveat": "QwenSWEBench is Qwen's software-engineering agent benchmark for repository-level issue resolution.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "QwenSWEBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 979, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwen-swe-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qwenwebbench", "caveat": "QwenWebBench is an internal front-end code generation benchmark by Qwen. It is bilingual (EN/CN) and spans 7 categories (Web Design, Web Apps, Games, SVG, Data Visualization, Animation, and 3D), using auto-render plus a multimodal judge for code and visual correctness. Scores are reported as BT/Elo ratings.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "QwenWebBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 980, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenwebbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-qwenworldbench", "caveat": "QwenWorldBench is Qwen's internal benchmark for evaluating LLMs as world models that simulate agentic environments across Terminal, SWE, MCP, Search, OS, Android, and Web domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "QwenWorldBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 981, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/qwenworldbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-516-race-high", "caveat": "RACE is a large-scale reading comprehension dataset with more than 28,000 passages and nearly 100,000 questions. The dataset is collected from English examinations in China, which are designed for middle school and high school students. RACE 是一个大规模的阅读理解数据集，包含超过 28,000 个段落和近 100,000 个问题。该数据集是从中国的英语考试中收集而来，这些考试是为中学和高中学生设计的。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RACE%28High%29"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "RACE(High)", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 982, "released": "2017-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28High%29"}, {"aliases": [], "benchmark_id": "opencompass-517-race-middle", "caveat": "RACE is a large-scale reading comprehension dataset with more than 28,000 passages and nearly 100,000 questions. The dataset is collected from English examinations in China, which are designed for middle school and high school students. RACE 是一个大规模的阅读理解数据集，包含超过 28,000 个段落和近 100,000 个问题。该数据集是从中国的英语考试中收集而来，这些考试是为中学和高中学生设计的。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RACE%28Middle%29"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "RACE(Middle)", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 983, "released": "2017-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RACE%28Middle%29"}, {"aliases": [], "benchmark_id": "opencompass-1918-rdb2g-bench", "caveat": "RDB2G-Bench provides comprehensive performance evaluation data for graph neural network models applied to relational database tasks. The dataset contains extensive experiments across multiple graph configurations and architectures. RDB2G-Bench提供了针对关系数据库任务应用的图神经网络模型的全面性能评估数据。该数据集包含了跨多种图配置和架构的广泛实验。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RDB2G-Bench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "RDB2G-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 984, "released": "2025-06-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RDB2G-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1252-re-bench", "caveat": "RE-Bench (Research Engineering Benchmark, v1) consists of 7 challenging, open-ended ML research engineering environments and data from 71 8-hour attempts by 61 distinct human experts. RE-Bench用于评估AI智能体研发的自动化能力，它由61位人类专家71次在7个具有挑战性的开放式ML研究工程环境中的8小时尝试的数据组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RE-Bench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "RE-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 985, "released": "2024-11-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RE-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1755-real", "caveat": "Benchmarking Autonomous Agents on Deterministic Simulations of Real Websites 在真实网站的确定性模拟上对自主代理进行基准测试", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/REAL"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "REAL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 986, "released": "2025-04-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/REAL"}, {"aliases": [], "benchmark_id": "llm-stats-realkie-fcc", "caveat": "RealKIE-FCC is a key information extraction benchmark drawn from real enterprise documents (FCC filings), part of the RealKIE suite of five novel datasets for enterprise key information extraction. Models must convert documents to markdown and extract structured fields against a specified JSON schema. Nova 2 reports results on a human-verified version of the dataset.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "RealKIE-FCC", "organization_count": 1, "organizations": ["llm_stats"], "rank": 987, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/realkie-fcc?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1124-realtoxicityprompts", "caveat": "RealToxicityPrompts is a dataset of 100K naturally occurring, sentence-level prompts derived from a large corpus of English web text, paired with toxicity scores from a widely used toxicity classiﬁer. RealToxicityPrompts 是一个包含 100,000 个自然出现的、句子级提示的数据集，这些提示来自于大量的英语网络文本，并配有来自广泛使用的毒性分类器的毒性评分。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RealToxicityPrompts"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "RealToxicityPrompts", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 988, "released": "2020-11-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RealToxicityPrompts"}, {"aliases": [], "benchmark_id": "llm-stats-realworldqa", "caveat": "RealWorldQA is a benchmark designed to evaluate basic real-world spatial understanding capabilities of multimodal models. The initial release consists of over 700 anonymized images taken from vehicles and other real-world scenarios, each accompanied by a question and easily verifiable answer. Released by xAI as part of their Grok-1.5 Vision preview to test models' ability to understand natural scenes and spatial relationships in everyday visual contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "RealWorldQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 989, "released": "2024-04-12", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/realworldqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1361-realworldqa", "caveat": "RealWorldQA is a benchmark designed for real-world understanding, including 765 images, each accompanied by a question and a verifiable answer. RealWorldQA用于评估多模态模型在现实世界中的空间理解能力，包含765张图像，每张图像都配有一个问题和易于验证的答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RealworldQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "RealworldQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 990, "released": "2024-04-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RealworldQA"}, {"aliases": [], "benchmark_id": "opencompass-530-record", "caveat": "ReCoRD is a reading comprehension task, which requires to extract the answer from the article based on the given news article and question. ReCoRD是一个阅读理解任务，要求根据给定的新闻文章和问题，从文章中抽取出答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ReCoRD"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "ReCoRD", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 991, "released": "2018-10-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ReCoRD"}, {"aliases": [], "benchmark_id": "llm-stats-recreationbench", "caveat": "RecreationBench is Qwen's long-horizon application-recreation benchmark for hybrid agents across Ubuntu, macOS, Windows, Android, and the web.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "RecreationBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 992, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/recreationbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1333-redcode", "caveat": "RedCode provides comprehensive and practical evaluations on the safety of code agents, including 4,050 risky test cases covering 25 types of critical vulnerabilities spanning 8 domains and 160 prompts aiming to generate harmful code or software. RedCode旨在为LLM代码智能体的安全性提供全面实用的评估，包括来自8个领域25种关键漏洞的4050个风险测试用例，以及160个生成有害代码的提示。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RedCode"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "RedCode", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 993, "released": "2024-11-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RedCode"}, {"aliases": [], "benchmark_id": "llm-stats-refcoco-avg", "caveat": "RefCOCO-avg measures object grounding accuracy averaged across RefCOCO, RefCOCO+, and RefCOCOg benchmarks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "RefCOCO-avg", "organization_count": 1, "organizations": ["llm_stats"], "rank": 994, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/refcoco-avg?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-refcocog", "caveat": "RefCOCOg is a referring expression comprehension benchmark that evaluates spatial grounding in images. Given a natural language expression describing an object, the model must localize the correct region, evaluated by accuracy at a 0.5 IoU threshold. It features longer, more descriptive expressions than RefCOCO and RefCOCO+.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "RefCOCOg", "organization_count": 1, "organizations": ["llm_stats"], "rank": 995, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/refcocog?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-refspatialbench", "caveat": "RefSpatialBench evaluates spatial reference understanding and grounding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "RefSpatialBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 996, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/refspatialbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1317-repliqa", "caveat": "RepLiQA is suited for question-answering and topic retrieval tasks, including collection of five splits of test sets. Accurate answers can only be generated if a model can find relevant content within the provided document. RepLiQA适用于问答和主题检索任务，集合了是5个测试集；只有当模型可以在提供的文档中找到相关内容时，才能生成准确的答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RepLiQA"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "RepLiQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 997, "released": "2024-06-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RepLiQA"}, {"aliases": [], "benchmark_id": "llm-stats-repo-env", "caveat": "Repo Env evaluates an agent's ability to set up, configure, and run real repositories, including dependency resolution and environment management.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Repo Env", "organization_count": 1, "organizations": ["llm_stats"], "rank": 998, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/repo-env?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-repobench", "caveat": "RepoBench is a benchmark for evaluating repository-level code auto-completion systems through three interconnected tasks: RepoBench-R (retrieval of relevant code snippets across files), RepoBench-C (code completion with cross-file and in-file context), and RepoBench-P (pipeline combining retrieval and prediction). Supports Python and Java programming languages and addresses the gap in evaluating real-world, multi-file programming scenarios by providing a more complete comparison of performance in auto-completion systems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "RepoBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 999, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/repobench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-repoqa", "caveat": "RepoQA is a benchmark for evaluating long-context code understanding capabilities of Large Language Models through the Searching Needle Function (SNF) task, where LLMs must locate specific functions in code repositories using natural language descriptions. The benchmark contains 500 code search tasks spanning 50 repositories across 5 modern programming languages (Python, Java, TypeScript, C++, and Rust), tested on 26 general and code-specific LLMs to assess their ability to comprehend and navigate code repositories.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "RepoQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1000, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/repoqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-researchclawbench", "caveat": "ResearchClawBench evaluates research agents on realistic, tool-using research tasks that require code execution and filesystem workspace interaction.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "research", "name": "ResearchClawBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1001, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/researchclawbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1087-reveal", "caveat": "REVEAL(ReasoningVerification Evaluation) is a new dataset to benchmark automatic verifiers of complex Chain-ofThought reasoning in open-domain question answering settings. Reveal 是一个用于基准测试开放域问答环境中复杂链式推理自动验证器的新数据集。Reveal 包含关于语言模型答案中每个推理步骤的相关性、证据段落的归因和逻辑正确性的全面标签，涵盖多种数据集和最先进的语言模型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Reveal"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "Reveal", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1002, "released": "2024-05-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Reveal"}, {"aliases": [], "benchmark_id": "opencompass-1915-rewardbench", "caveat": "RewardBench is a benchmark designed to evaluate the capabilities and safety of reward models (including those trained with Direct Preference Optimization, DPO). RewardBench是一个用于评估奖励模型（包括通过直接偏好优化（DPO）训练的模型）能力和安全性的基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RewardBench"], "document_share": 0.0008271298593879239, "domain": "指令跟随", "name": "RewardBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1003, "released": "2025-06-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RewardBench"}, {"aliases": [], "benchmark_id": "opencompass-2033-rexbench", "caveat": "RExBench is a benchmark for evaluating large language model agents’ ability to autonomously implement AI research extensions, focusing on code generation, experimental design, and research comprehension. RExBench 是一个评估大型语言模型代理在自动实现 AI 研究扩展能力的基准，涵盖代码生成、实验设计和研究理解等维度。该基准包含12个基于真实论文和代码库的任务，每项任务由领域专家提供扩展指令，并通过自动化基础设施执行代理输出以验证成功标准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RExBench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "RExBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1004, "released": "2025-06-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RExBench"}, {"aliases": [], "benchmark_id": "opencompass-1656-rfuav", "caveat": "RFUAV offers a comprehensive benchmark dataset for Radio-Frequency (RF)-based drone detection and identification. RFUAV 提供了一个基于射频（RF）的无人机检测和识别的全面基准数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RFUAV"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "RFUAV", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1005, "released": "2025-03-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RFUAV"}, {"aliases": [], "benchmark_id": "opencompass-2371-rigorousbench", "caveat": "Artificial intelligence is undergoing the paradigm shift from closed language models to interconnected agent systems capable of external perception and information integration. As a representative embodiment, Deep Research Agents (DRAs) systematically exhibit the capabilities for task decomposition, Artificial intelligence is undergoing the paradigm shift from closed language models to interconnected agent systems capable of external perception and information integration. As a representative embodiment, Deep Research Agents (DRAs) systematically exhibit the capabilities for task decomposition,", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RigorousBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "RigorousBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1006, "released": "2025-10-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RigorousBench"}, {"aliases": [], "benchmark_id": "opencompass-2048-risebench", "caveat": "RISEBench is a benchmark for evaluating large multimodal models (LMMs) on reasoning-informed visual editing tasks, targeting models with image understanding and generation capabilities. RISEBench 是一个用于评估多模态大模型（LMMs）在推理驱动视觉编辑任务中能力的基准，面向具备图像理解与生成能力的模型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RISEBench"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "RISEBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1007, "released": "2025-04-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RISEBench"}, {"aliases": [], "benchmark_id": "opencompass-1524-rm-bench", "caveat": "RM-Bench, a benchmark dataset for evaluating reward models of language modeling. RM-Bench，一个用于评估语言模型奖励模型的基准数据集", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RM-Bench"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "RM-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1008, "released": "2024-10-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RM-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-robospatialhome", "caveat": "RoboSpatialHome evaluates spatial understanding for robotic home navigation and manipulation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500"], "document_share": 0.0008271298593879239, "domain": "robotics", "name": "RoboSpatialHome", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1009, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/robospatialhome?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-robust-if", "caveat": "Robust IF evaluates instruction-following robustness on diverse, hard prompts, measuring whether a model reliably adheres to constraints across challenging single-turn and multi-turn scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Robust IF", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1010, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/robust-if?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1091-rolellm", "caveat": "RoleLLM is a role-playing framework of data construction and evaluation (RoleBench), as well as solutions for both closed-source and open-source models (RoleGPT, RoleLLaMA, RoleGLM). We also propose Context-Instruct for long-text knowledge extraction and role-specific knowledge injection. RoleLLM 是一个角色扮演的数据构建和评估框架，同时提供闭源和开源模型的解决方案（RoleGPT、RoleLLaMA、RoleGLM）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RoleLLM"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "RoleLLM", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1011, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RoleLLM"}, {"aliases": [], "benchmark_id": "llm-stats-rsi-index", "caveat": "The RSI (Recursive Self-Improvement) Index is OpenAI's aggregate metric across a bundle of internal AI-research evaluations, including debugging research systems, optimizing kernels and training recipes, and improving other models, measuring progress toward recursive self-improvement.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "RSI Index", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1012, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/rsi-index?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1678-rsmmvp", "caveat": "This dataset follows a similar procedure to the original MMVP benchmark on natural images but directed towards the remote sensing domain. Challenging visual patterns are identified based on CLIP blind pairs, accompanied with the correpsonding questions, options and ground-truth answer. RSMMVP遵循与原始 MMVP 基准在自然图像上的类似流程，但针对遥感领域。根据 CLIP 盲对识别具有挑战性的视觉模式，并附带相应的问题、选项和真实答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RSMMVP"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "RSMMVP", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1013, "released": "2025-03-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RSMMVP"}, {"aliases": [], "benchmark_id": "opencompass-528-rte", "caveat": "RTE is a natural language inference task, which requires to determine the logical relation between the given sentence pair, with three relations: entailment, contradiction and neutral. RTE是一个自然语言推理任务，要求根据给定的句子对，判断它们之间的逻辑关系，有三种关系：蕴含、矛盾和中立。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RTE"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "RTE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1014, "released": null, "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RTE"}, {"aliases": [], "benchmark_id": "llm-stats-ruler", "caveat": "RULER v1 is a synthetic long-context benchmark for measuring how model quality degrades as input length increases. This packaging follows the public standalone NVIDIA RULER implementation with 13 official tasks spanning retrieval, multi-hop tracing, aggregation, and QA.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "RULER", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1015, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ruler-1000k", "caveat": "RULER 1000K evaluates the official 13-task RULER v1 suite at a 1048576-token (1M) context budget.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "RULER 1000K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1016, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-1000k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ruler-128k", "caveat": "RULER 128k evaluates the official 13-task RULER v1 suite at a 131072-token context budget.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "RULER 128k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1017, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-128k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ruler-2048k", "caveat": "RULER 2048K evaluates the official 13-task RULER v1 suite at a 2097152-token (2M) context budget.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "RULER 2048K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1018, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-2048k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ruler-512k", "caveat": "RULER 512K evaluates the official 13-task RULER v1 suite at a 524288-token context budget.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "RULER 512K", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1019, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-512k?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-ruler-64k", "caveat": "RULER 64k evaluates the official 13-task RULER v1 suite at a 65536-token context budget.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "RULER 64k", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1020, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/ruler-64k?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1738-rulistening", "caveat": "RUListening: Robust Understanding through Listening, an automated QA generation framework for evaluating multimodal perception. RUListening：通过聆听的稳健理解，这是一个用于评估多模态感知的自动问答生成框架。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RUListening"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "RUListening", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1021, "released": "2025-04-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RUListening"}, {"aliases": [], "benchmark_id": "opencompass-1707-rxrx3-core", "caveat": "RxRx3-core dataset is a challenge dataset in phenomics optimized for the research\ncommunity. RxRx3-core数据集是Recursion为研究社区优化的表型组学挑战数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/RXRX3-CORE"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "RXRX3-CORE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1022, "released": "2025-03-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/RXRX3-CORE"}, {"aliases": [], "benchmark_id": "opencompass-1003-s-eval", "caveat": "S-Eval is a new comprehensive, multi-dimensional and open-ended safety evaluation benchmark for LLMs consisting of 220,000 evaluation prompts (still in active expansion) across 102 risk subcategories and 10 advanced jailbreak attacks. S-Eval 是一个针对 LLM 的全新全面、多维、开放式安全评估基准，包含 102 个风险子类别的 220,000 个评估提示（仍在积极扩展中）和 10 个高级越狱攻击。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/S-Eval"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "S-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1023, "released": "2024-05-23", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/S-Eval"}, {"aliases": [], "benchmark_id": "opencompass-1772-s1-bench", "caveat": "S1-Bench is a novel benchmark designed to evaluate the performance of LRMs on simple tasks that are more aligned with intuitive System 1 thinking, rather than deliberate System 2 reasoning. S1-Bench offers a set of simple, diverse, and naturally clear questions across multiple domains and languages. S1-Bench是一个新颖的基准，旨在评估大模型在简单任务中的表现，这些任务更倾向于直观的系统1思维，而非深思熟虑的系统2推理。尽管大模型在复杂推理任务中通过明确的思维链取得了显著突破，但它们对深度分析思维的依赖可能限制了其系统1思维能力。此外，目前缺乏评估大模型在需要此类能力的任务中表现的基准。为了填补这一空白，S1-Bench提供了一组简单、多样且自然清晰的问题，涵盖多个领域和语言，专门设计用于评估大模型在此类任务中的表现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/S1-Bench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "S1-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1024, "released": "2025-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/S1-Bench"}, {"aliases": [], "benchmark_id": "opencompass-2083-saebench", "caveat": "SAEBench is a comprehensive benchmark designed to evaluate and compare the performance of language model sparse autoencoders (SAEs). This benchmark provides over 200 SAE models covering seven architectures for systematic comparison, and has open-source code and models. SAEBench 是一个旨在评估和比较语言模型稀疏自动编码器（SAEs）性能的综合性基准。该基准提供超过200个SAE模型，涵盖七种架构，以实现系统性比较，并且开源了代码和模型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SAEBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "SAEBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1025, "released": "2025-03-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SAEBench"}, {"aliases": [], "benchmark_id": "opencompass-1073-safetybench", "caveat": "SafetyBench, a comprehensive benchmark for evaluating the safety of LLMs, which comprises 11,435 diverse multiple choice questions spanning across 7 distinct categories of safety concerns. Notably, SafetyBench also incorporates both Chinese and English data. SafetyBench 是一个全面的基准，用于评估大型语言模型（LLMs）的安全性，包含 11,435 道多样化的选择题，涵盖 7 个不同的安全关注类别。SafetyBench 还包含中文和英文的数据，方便双语评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SafetyBench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "SafetyBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1026, "released": "2024-06-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SafetyBench"}, {"aliases": [], "benchmark_id": "opencompass-1077-salad-bench", "caveat": "SALAD-Bench, a safety benchmark specifically designed for evaluating LLMs, attack, and defense methods. Distinguished by its breadth, SALAD-Bench transcends conventional benchmarks through its large scale, rich diversity, intricate taxonomy\nspanning three levels, and versatile functionalities. SALAD-Bench 是一个专门用于评估大型语言模型（LLMs）、攻击和防御方法的安全基准。SALAD-Bench 的特点在于其广泛性，超越了传统基准，具有大规模、丰富的多样性、复杂的三层分类法以及多功能性。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SALAD-Bench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "SALAD-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1027, "released": "2024-02-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SALAD-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-sat-math", "caveat": "SAT Math benchmark from AGIEval containing standardized mathematics questions from the College Board SAT examination, designed to evaluate mathematical reasoning capabilities of foundation models using human-centric assessment methods.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "SAT Math", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1028, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/sat-math?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1732-scam", "caveat": "SCAM is the largest and most diverse dataset of real-world typographic attack images to date, containing 1,162 images across hundreds of object categories and attack words. SCAM，是迄今为止规模最大、多样性最丰富的真实世界排版攻击图像数据集，包含数百个对象类别和攻击词汇的1,162张图像。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SCAM"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SCAM", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1029, "released": "2025-04-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SCAM"}, {"aliases": [], "benchmark_id": "artificial-analysis-scicode", "caveat": "Coding", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/scicode"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "SciCode", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 1030, "released": "2024-07-18", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/scicode"}, {"aliases": [], "benchmark_id": "llm-stats-scicode", "caveat": "SciCode is a research coding benchmark curated by scientists that challenges language models to code solutions for scientific problems. It contains 338 subproblems decomposed from 80 challenging main problems across 16 natural science sub-fields including mathematics, physics, chemistry, biology, and materials science. Problems require knowledge recall, reasoning, and code synthesis skills.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "SciCode", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1031, "released": "2024-07-18", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/scicode?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-scienceqa", "caveat": "ScienceQA is the first large-scale multimodal science question answering benchmark with 21,208 multiple-choice questions covering 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. The benchmark includes both text and image modalities, featuring detailed explanations and Chain-of-Thought reasoning to diagnose multi-hop reasoning ability.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "ScienceQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1032, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1100-scienceqa", "caveat": "SCIENCEQA is a new benchmark that consists of ∼21k multimodal multiple choice questions with diverse science topics and annotations of their answers with corresponding lectures and explanations. SCIENCEQA 包含约 21,000 道多模态选择题，涵盖多种科学主题，并附有相应的讲座和解释的答案注释。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ScienceQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "ScienceQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1033, "released": "2022-10-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ScienceQA"}, {"aliases": [], "benchmark_id": "llm-stats-scienceqa-visual", "caveat": "ScienceQA Visual is a multimodal science question answering benchmark consisting of 21,208 multiple-choice questions from elementary and high school science curricula. The dataset covers 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. 48.7% of questions include image context requiring multimodal reasoning. Questions are annotated with lectures (83.9%) and explanations (90.5%) to support chain-of-thought reasoning for science question answering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ScienceQA Visual", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1034, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/scienceqa-visual?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1323-scifibench", "caveat": "SciFIBench is a scientific figure interpretation benchmark for LMMs, consisting of 2000 questions split between two tasks across 8 categories. SciFIBench用于评估多模态大模型的科学图表解释能力，由2000个问题组成，涵盖8个类别的2种任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SciFIBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SciFIBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1035, "released": "2024-05-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SciFIBench"}, {"aliases": [], "benchmark_id": "llm-stats-screenspot", "caveat": "ScreenSpot is the first realistic GUI grounding benchmark that encompasses mobile, desktop, and web environments. The dataset comprises over 1,200 instructions from iOS, Android, macOS, Windows and Web environments, along with annotated element types (text and icon/widget), designed to evaluate visual GUI agents' ability to accurately locate screen elements based on natural language instructions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ScreenSpot", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1036, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-screenspot-pro", "caveat": "ScreenSpot-Pro is a novel GUI grounding benchmark designed to rigorously evaluate the grounding capabilities of multimodal large language models (MLLMs) in professional high-resolution computing environments. The benchmark comprises 1,581 instructions across 23 applications spanning 5 industries and 3 operating systems, featuring authentic high-resolution images from professional domains with expert annotations. Unlike previous benchmarks that focus on cropped screenshots in consumer applications, ScreenSpot-Pro addresses the complexity and diversity of real-world professional software scenarios, revealing significant performance gaps in current MLLM GUI perception capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ScreenSpot Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1037, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/screenspot-pro?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-seal-0", "caveat": "Seal-0 is a benchmark for evaluating agentic search capabilities, testing models' ability to navigate and retrieve information using tools.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Seal-0", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1038, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/seal-0?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-openai-search-function-calling", "caveat": "Search and Function-Calling is an OpenAI internal production benchmark measuring reliable search-tool use and function calling in agentic workflows, reported as a pass rate.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Search and Function-Calling", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1039, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/openai-search-function-calling?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1986-sec-bench", "caveat": "SEC-bench is a benchmark designed to evaluate large language model (LLM) agents on real-world software security tasks. SEC-bench 是一个面向大型语言模型（LLM）智能体的软件安全任务评测基准，旨在自动化评估模型在真实漏洞环境中的能力。 该基准通过多智能体框架自动构建代码仓库、复现漏洞并生成修复补丁，涵盖漏洞验证（PoC 生成）和补丁修复两个关键任务，包含数百个真实安全案例。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEC-bench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "SEC-bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1040, "released": "2025-06-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SEC-bench"}, {"aliases": [], "benchmark_id": "llm-stats-sec-bench-pro", "caveat": "SEC-bench Pro is a self-evolving software-security benchmark that measures agent bug hunting on critical, high-complexity systems. It instantiates validated vulnerabilities across the V8 and SpiderMonkey JavaScript engines as reproducible vulnerability-discovery and proof-of-concept-generation tasks with oracle-based validation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "SEC-bench Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1041, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/sec-bench-pro?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-948-secbench", "caveat": "Tencent Zhuque Lab and Tencent Security Keen Lab, together with Tencent Huyuan Team, Professor Jiang Yong/Professor Xia Shutao's team from Tsinghua University, Professor Luo Xiapu's research team from Hong Kong Polytechnic University, and OpenCompass team from Shanghai Artificial Intelligence Laboratory, have jointly built a safety benchmark, SecBench. We provide fair, impartial, objective, and comprehensive evaluation capabilities and promote the construction of large model in security dimension. 腾讯朱雀实验室和腾讯安全科恩实验室联合腾讯混元大模型团队、清华大学江勇教授/夏树涛教授团队、香港理工大学罗夏朴教授研究团队以及上海人工智能实验室OpenCompass团队，通过建设安全大模型评测基准SecBench，为安全大模型研发提供公平、公正、客观、全面的评测能力，推动安全大模型建设。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SecBench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "SecBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1042, "released": "2024-01-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SecBench"}, {"aliases": [], "benchmark_id": "llm-stats-seccodebench", "caveat": "SecCodeBench evaluates LLM coding agents on secure code generation and vulnerability detection, testing the ability to produce code that is both functional and free from security vulnerabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "SecCodeBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1043, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/seccodebench?top_n=500"}, {"aliases": ["SecCodeBench"], "benchmark_id": "seccodebench", "caveat": "Measures secure-coding behaviour, not exploitation capability.", "document_count": 1, "document_ids": ["model_reports:qwen3_5_model_card"], "document_share": 0.0008271298593879239, "domain": "security", "name": "SecCodeBench", "organization_count": 1, "organizations": ["Qwen"], "rank": 1044, "released": "2025-10-14", "source": "model_reports", "url": "https://github.com/alibaba/SecCodeBench"}, {"aliases": [], "benchmark_id": "opencompass-1359-seed-bench", "caveat": "SEED-Bench aims at the evaluation of generative comprehension in MLLMs, consisting of 19K multiple choice questions, which spans 12 evaluation dimensions including the comprehension of both the image and video modality. SEED-Bench用于评估多模态大模型的理解能力，包括对图像和视频的理解，由跨越12个评估维度的19K道多项选择题组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEED-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SEED-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1045, "released": "2023-07-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench"}, {"aliases": [], "benchmark_id": "opencompass-1363-seed-bench-2", "caveat": "SEED-Bench-2 assesses both text and image generation of MLLMs. It spans 27 evaluation dimensions, featuring 24K multiple-choice questions with precise human annotations. SEED-Bench-2用于评估多模态大模型的文本和图像生成能力，包括跨越27个维度的24K道多选题及准确的人工注释。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SEED-Bench-2", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1046, "released": "2023-11-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2"}, {"aliases": [], "benchmark_id": "opencompass-1395-seed-bench-2-plus", "caveat": "SEED-Bench-2-Plus, a benchmark specifically designed for evaluating text-rich visual comprehension of MLLMs. The benchmark comprises 2.3K multiple-choice questions with precise human annotations, spanning three broad categories: Charts, Maps, and Webs. SEED-Bench-2-Plus, a benchmark specifically designed for evaluating text-rich visual comprehension of MLLMs. The benchmark comprises 2.3K multiple-choice questions with precise human annotations, spanning three broad categories: Charts, Maps, and Webs.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2-Plus"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SEED-Bench-2-Plus", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1047, "released": "2024-04-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SEED-Bench-2-Plus"}, {"aliases": [], "benchmark_id": "llm-stats-seedclawbench", "caveat": "SeedClawBench is an agentic coding benchmark measuring overall model performance on real-world, tool-using software development tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "SeedClawBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1048, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/seedclawbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1961-sfe", "caveat": "The Scientists' First Exam (SFE) benchmark, designed to comprehensively evaluate the scientific cognitive capabilities of MLLMs through three cognitive levels (cog-levels):Scientific Signal Perception、Scientific Attribute Understanding 、Scientific Comparative Reasoning. The Scientists' First Exam (SFE) 基准测试旨在通过三个认知层级——科学信号感知、科学属性理解 和 科学对比推理，全面评估多模态大语言模型（MLLMs）的科学认知能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SFE"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SFE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1049, "released": "2025-06-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SFE"}, {"aliases": [], "benchmark_id": "opencompass-1330-sg-bench", "caveat": "SG-Bench assess LLM safety across various tasks and prompt types. It integrates both generative and discriminative evaluation tasks and includes extended data to examine the impact of prompt engineering and jailbreak. SG-Bench用于评估LLM在不同任务和提示下的安全性，整合了生成性和判别性评估任务，并包含扩展数据以度量提示工程和越狱对安全性的影响。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SG-Bench"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "SG-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1050, "released": "2024-10-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SG-Bench"}, {"aliases": [], "benchmark_id": "opencompass-2325-sgi-bench", "caveat": "SGI-Bench operationalizes this definition via four scientist-aligned task families: deep research, idea generation, dry/wet experiments, and multimodal experimental reasoning. The benchmark spans 10 disciplines and more than 1,000 expert-curated samples inspired by Science's 125 Big Questions. 科学通用智能（SGI）被定义为一种人工智能系统，其能够以接近人类科学家的通用性与熟练程度，自主地贯穿并迭代完整的科学研究流程，包括审思、构思、行动与感知四个阶段。SGI-Bench 通过四类与科学家工作流程对齐的任务对上述定义进行操作化刻画，具体包括：深度研究、思想与假设生成、干/湿实验、以及多模态实验推理。该评测基准覆盖10 个科学学科领域，包含1,000 余个由领域专家精心策划的样本，其设计灵感来源于 Science 杂志提出的 125 个重大科学问题（125 Big Questions）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SGI-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SGI-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1051, "released": "2025-12-22", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SGI-Bench"}, {"aliases": [], "benchmark_id": "opencompass-2348-shell", "caveat": "Targeting overlooked implicit risks in vertical domains (Education, Finance, Management), we introduce an implicit risk benchmark and the MENTOR framework. By leveraging Rule Evolution and Activation Steering, MENTOR effectively detects and mitigates these subtle hazards. Shell由华东师范大学Shell@Educhat团队和上海人工智能实验室联合推出。当下，确保垂直领域任务中大模型的安全性至关重要。虽然目前的对齐工作主要针对偏见和暴力等显性风险，但往往忽略了更深层次的特定领域隐性风险。研发团队推出了一个包含大量隐式风险查询的基准测试集，将风险分为引导、反思、禁止三类，以及 MENTOR 框架。该框架利用规则演化循环（REC）和激活引导（RV）技术，能够有效发现并缓解这些不易察觉的潜在风险。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Shell"], "document_share": 0.0008271298593879239, "domain": "安全", "name": "Shell", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1052, "released": "2025-12-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Shell"}, {"aliases": [], "benchmark_id": "opencompass-1321-shoppingmmlu", "caveat": "Shopping MMLU is a diverse multi-task online shopping benchmark derived from real-world Amazon data. It consists of 57 tasks covering 4 major shopping skills: concept understanding, knowledge reasoning, user behavior alignment, and multi-linguality. Shopping MMLU是一个基于真实亚马逊数据的多样化多任务在线购物基准测试，由57项任务组成，涵盖概念理解、知识推理、用户行为对齐和多语言4大购物场景技能。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ShoppingMMLU"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "ShoppingMMLU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1053, "released": "2024-10-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ShoppingMMLU"}, {"aliases": [], "benchmark_id": "llm-stats-sifo", "caveat": "SIFO (Simple Instruction Following) evaluates how well language models follow simple, explicit instructions. It tests fundamental instruction-following capabilities across various task types.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500"], "document_share": 0.0008271298593879239, "domain": "structured_output", "name": "SIFO", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1054, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-sifo-multiturn", "caveat": "SIFO-Multiturn evaluates instruction following capabilities in multi-turn conversational settings, testing how well models maintain context and follow instructions across multiple exchanges.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500"], "document_share": 0.0008271298593879239, "domain": "structured_output", "name": "SIFO-Multiturn", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1055, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/sifo-multiturn?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-simpleqa", "caveat": "SimpleQA is a factuality benchmark developed by OpenAI that measures the short-form factual accuracy of large language models. The benchmark contains 4,326 short, fact-seeking questions that are adversarially collected and designed to have single, indisputable answers. Questions cover diverse topics from science and technology to entertainment, and the benchmark also measures model calibration by evaluating whether models know what they know.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SimpleQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1056, "released": "2024-10-30", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-simpleqa-verified", "caveat": "SimpleQA Verified is a curated, reliability-focused subset of SimpleQA that addresses label noise and redundancy in the original benchmark, measuring short-form parametric factual accuracy of large language models on fact-seeking questions with single, indisputable answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SimpleQA Verified", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1057, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/simpleqa-verified?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-simplevqa", "caveat": "SimpleVQA is a visual question answering benchmark focused on simple queries.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "SimpleVQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1058, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/simplevqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-533-siqa", "caveat": "SIQA is a social interaction question answering task, which requires to select the most reasonable behavior based on the given scenario and three possible subsequent behaviors. This task is designed to test the model's knowledge in social commonsense. This dataset consists of 38,963 training samples, 1,951 development samples and 1,960 test samples, all on English text. SIQA 是一个社会交互问答任务，要求根据给定的场景和三个可能的后续行为，选择最合理的行为。这个任务是为了测试模型在社会常识方面的知识。这个数据集包含了 38963 个训练样本，1951 个开发样本和 1960 个测试样本，所有的文本都是英文文本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SIQA"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "SIQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1059, "released": "2019-09-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SIQA"}, {"aliases": [], "benchmark_id": "llm-stats-siren-agentdojo-attack-success", "caveat": "Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric is the attack success rate; lower is better.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "Siren AgentDojo Attack Success Rate", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1060, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-attack-success?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-siren-agentdojo-utility", "caveat": "Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric reports utility on the assigned tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "Siren AgentDojo Utility", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1061, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/siren-agentdojo-utility?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-skillsbench", "caveat": "SkillsBench evaluates coding agents on self-contained programming tasks, measuring practical engineering skills across diverse software development scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "SkillsBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1062, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/skillsbench?top_n=500"}, {"aliases": ["SkillsBench", "SkillsBench V1", "Skills Bench"], "benchmark_id": "skillsbench", "caveat": "Measures whether curated Agent Skills raise agent pass rates, not raw model capability; the meaningful figure is the paired no-Skills versus curated-Skills gap. The release date follows the paper's first submission.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "agent", "name": "SkillsBench", "organization_count": 1, "organizations": ["Tencent"], "rank": 1063, "released": "2026-02-13", "source": "model_reports", "url": "https://skillsbench.ai/"}, {"aliases": [], "benchmark_id": "llm-stats-slakevqa", "caveat": "A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "SlakeVQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1064, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/slakevqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2217-smartbench", "caveat": "SmartBench is the first benchmark specifically designed to evaluate the capabilities of on-device large language models in smartphone scenarios. SmartBench 是首个面向智能手机终端大模型能力评估的基准，基于手机厂商提供的功能将其划分为五类共20项任务，涵盖文本摘要、问答、信息抽取、内容创作和通知管理等场景。它提供高质量数据集与定制化评估标准，旨在推动终端大模型在移动应用中的标准化评估与发展。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SmartBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SmartBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1065, "released": "2025-09-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SmartBench"}, {"aliases": [], "benchmark_id": "opencompass-2029-smmile", "caveat": "SMMILE is a benchmark for evaluating multimodal large language models (MLLMs) in medical in-context learning tasks, constructed by medical experts. SMMILE 是一个由医学专家主导构建的多模态医疗上下文学习评测基准，旨在评估多模态大语言模型（MLLMs）在医学任务中的上下文学习能力。该基准包含111个问题（共517个图文问答三元组），涵盖6个医学专科和13种影像模态，另提供扩展版本 SMMILE++，包含1038个变换问题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SMMILE"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "SMMILE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1066, "released": "2025-06-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SMMILE"}, {"aliases": [], "benchmark_id": "llm-stats-social-iqa", "caveat": "The first large-scale benchmark for commonsense reasoning about social situations. Contains 38,000 multiple choice questions probing emotional and social intelligence in everyday situations, testing commonsense understanding of social interactions and theory of mind reasoning about the implied emotions and behavior of others.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "psychology", "name": "Social IQa", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1067, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/social-iqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2092-spatial457", "caveat": "Spatial457 systematically introduces four key capabilities—multi-object understanding, 2D and 3D localization, and 3D orientation—across five difficulty levels and seven question types, from simple object recognition to complex 6DoF spatial reasoning tasks. Spatial457 聚焦于空间推理的四项核心能力：多物体理解、二维位置识别、三维位置识别，以及三维朝向判断。这些能力对现实世界中复杂场景的认知和理解至关重要。我们构建了一种层级递进的评估结构，将问题划分为 7 类问题类型，覆盖从基础的单物体识别任务，到我们首次提出的更具挑战性的 6DoF 空间推理任务。整个数据集按 5 个难度等级组织，帮助系统性地评估模型在不同层次空间理解任务中的表现。\n\nSpatial457 不仅提升了空间推理评测的覆盖面和挑战性，也为后续研究提供了统一的基准，有助于推动多模态模型在真实世界空间认知任务中的发展。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Spatial457"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "Spatial457", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1068, "released": "2025-04-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Spatial457"}, {"aliases": [], "benchmark_id": "llm-stats-spider", "caveat": "A large-scale, complex and cross-domain semantic parsing and text-to-SQL dataset annotated by 11 college students. Contains 10,181 questions and 5,693 unique complex SQL queries on 200 databases with multiple tables, covering 138 different domains. Requires models to generalize to both new SQL queries and new database schemas, making it distinct from previous semantic parsing tasks that use single databases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Spider", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1069, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/spider?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1133-spider", "caveat": "Spider is a large-scale complex and cross-domain semantic parsing and text-to-SQL dataset annotated by 11 Yale students. The goal of the Spider challenge is to develop natural language interfaces to cross-domain databases. Spider 是一个大规模、复杂且跨领域的语义解析和文本到 SQL 数据集。Spider 挑战的目标是开发自然语言接口以访问跨领域数据库。该数据集包含 10,181 个问题和 5,693 个独特的复杂 SQL 查询，涵盖 200 个包含多个表的数据库，涉及 138 个不同的领域。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Spider"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "Spider", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1070, "released": "2019-02-02", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Spider"}, {"aliases": [], "benchmark_id": "opencompass-1270-spider2-v", "caveat": "Spider2-V is the first multimodal agent benchmark focusing on professional data science and engineering workflows, featuring 494 real-world tasks in authentic computer environments and incorporating 20 enterprise-level professional applications. Spider2-V是第一个专注于专业数据科学和工程工作流程的多模态代理基准测试，整合了20 个企业级专业应用程序，包含来自真实计算机环境的494 个真实任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Spider2-V"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "Spider2-V", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1071, "released": "2024-07-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Spider2-V"}, {"aliases": [], "benchmark_id": "opencompass-1152-sportqa", "caveat": "SportQA aims to evaluate LLMs in the context of sports understanding. SportQA encompasses over 70,000 multiple-choice questions across three distinct difficulty levels, each targeting different aspects of sports knowledge from basic historical facts to intricate, scenariobased reasoning tasks. SportQA 专门用于评估大型语言模型（LLMs）在体育理解方面的能力。SportQA 包含超过 70,000 道多项选择题，分为三个不同的难度级别，针对从基本历史事实到复杂情境推理任务的各种体育知识。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SportQA"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "SportQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1072, "released": "2024-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SportQA"}, {"aliases": [], "benchmark_id": "opencompass-1267-spreadsheetbench", "caveat": "SpreadsheetBench is a challenging spreadsheet manipulation benchmark built from 912 real questions gathered from online Excel forums. SpreadsheetBench是一个具有挑战性的电子表格操作基准测试，包含912个来自在线Excel论坛的真实问题", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SpreadsheetBench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "SpreadsheetBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1073, "released": "2024-06-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SpreadsheetBench"}, {"aliases": [], "benchmark_id": "llm-stats-spreadsheetbench-2", "caveat": "SpreadsheetBench 2 evaluates office automation agents on spreadsheet analysis, reasoning, and manipulation tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "productivity", "name": "SpreadsheetBench 2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1074, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-spreadsheetbench-v1", "caveat": "SpreadSheetBench-v1 evaluates office automation agents on spreadsheet reasoning and manipulation tasks, measuring the ability to analyze, transform, and operate on spreadsheet data through tools.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500"], "document_share": 0.0008271298593879239, "domain": "productivity", "name": "SpreadSheetBench-v1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1075, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/spreadsheetbench-v1?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-squality", "caveat": "SQuALITY (Summarization-format QUestion Answering with Long Input Texts, Yes!) is a long-document summarization dataset built by hiring highly-qualified contractors to read public-domain short stories (3000-6000 words) and write original summaries from scratch. Each document has five summaries: one overview and four question-focused summaries. Designed to address limitations in existing summarization datasets by providing high-quality, faithful summaries.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "SQuALITY", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1076, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/squality?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1084-stabletoolbench", "caveat": "StableToolBench, a benchmark evolving from ToolBench, proposing a virtual API server and stable evaluation system. The virtual API server contains a caching system and API simulators which are complementary to alleviate the change in API status. StableToolBench 是一个从 ToolBench 发展而来的基准，提出了一个虚拟 API 服务器和稳定的评估系统。虚拟 API 服务器包含一个缓存系统和 API 模拟器，这些组件相辅相成，以缓解 API 状态变化带来的影响。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StableToolBench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "StableToolBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1077, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/StableToolBench"}, {"aliases": [], "benchmark_id": "opencompass-1845-stark-10k", "caveat": "STARK is a comprehensive benchmark designed to systematically evaluate large language models (LLMs) and large reasoning models (LRMs) on spatiotemporal reasoning tasks, particularly for applications in cyber-physical systems (CPS) such as robotics, autonomous vehicles, and smart city infrastructure. STARK 是一个全面的基准测试套件，旨在系统评估大语言模型（LLMs）和大推理模型（LRMs）在时空推理任务中的表现，特别是在网络物理系统（CPS）中的应用，如机器人、自动驾驶和智能城市基础设施。该基准包含 26 种不同的时空任务，涵盖状态估计、时空关系推理和世界知识感知推理三个层次。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/STARK_10k"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "STARK_10k", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1078, "released": "2025-05-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/STARK_10k"}, {"aliases": [], "benchmark_id": "llm-stats-stem", "caveat": "A comprehensive multimodal benchmark dataset with 448 skills and 1,073,146 questions spanning all STEM subjects (Science, Technology, Engineering, Mathematics), designed to test neural models' vision-language STEM skills based on K-12 curriculum. Unlike existing datasets that focus on expert-level ability, this dataset includes fundamental skills designed around educational standards.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "STEM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1079, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/stem?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1107-strategyqa", "caveat": "STRATEGYQA is a question answering (QA) benchmark where the required reasoning steps are implicit in the question, and should be inferred using a strategy STRATEGYQA 是一个问答基准，其中所需的推理步骤在问题中是隐含的，可以通过策略进行推断。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StrategyQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "StrategyQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1080, "released": "2021-01-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/StrategyQA"}, {"aliases": [], "benchmark_id": "opencompass-1542-structflowbench", "caveat": "StructFlowBench is a structurally annotated multi-turn benchmark that leverages a structure-driven generation paradigm to enhance the simulation of complex dialogue scenarios. StructFlowBench，这是一个包含155条数据的结构化标注多轮基准，它利用结构驱动生成范式来增强复杂对话场景的模拟。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StructFlowBench"], "document_share": 0.0008271298593879239, "domain": "指令跟随", "name": "StructFlowBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1081, "released": "2025-02-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/StructFlowBench"}, {"aliases": [], "benchmark_id": "opencompass-2082-structtokenbench", "caveat": "StructTokenBench is a benchmark framework designed to comprehensively evaluate the quality and efficiency of protein structure tokenization methods, particularly focusing on fine-grained local substructures. StructTokenBench 是一个评估蛋白质结构标记化方法质量效率的基准。它关注细粒度局部子结构，评测维度包括下游有效性、敏感性、独特性和码本利用效率。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StructTokenBench"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "StructTokenBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1082, "released": "2025-02-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/StructTokenBench"}, {"aliases": [], "benchmark_id": "opencompass-1082-studenteval", "caveat": "STUDENTEVAL contains 1,749 prompts written by 80 students who have only completed one introductory Python course. STUDENTEVAL contains numerous non-expert prompts describing the same problem, enabling exploration of key factors in prompt success. StudentEval 包含 1,749 个由 80 名仅完成一门入门 Python 课程的学生撰写的提示。StudentEval 中包含许多非专家提示，描述相同的问题，使得探索提示成功的关键因素成为可能。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StudentEval"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "StudentEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1083, "released": "2024-08-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/StudentEval"}, {"aliases": [], "benchmark_id": "opencompass-1744-stylerec", "caveat": "A Benchmark Dataset for Prompt Recovery in Writing Style Transformation. 基于写作风格转换的提示词恢复的评测集", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/StyleRec"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "StyleRec", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1084, "released": "2025-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/StyleRec"}, {"aliases": [], "benchmark_id": "llm-stats-summscreenfd", "caveat": "SummScreenFD is the ForeverDreaming subset of the SummScreen dataset for abstractive screenplay summarization, comprising pairs of TV series transcripts and human-written recaps from 88 different shows. The dataset provides a challenging testbed for abstractive summarization where plot details are often expressed indirectly in character dialogues and scattered across the entirety of the transcript, requiring models to find and integrate these details to form succinct plot descriptions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500"], "document_share": 0.0008271298593879239, "domain": "long_context", "name": "SummScreenFD", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1085, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/summscreenfd?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-sunrgbd", "caveat": "SUNRGBD evaluates RGB-D scene understanding and 3D grounding capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500"], "document_share": 0.0008271298593879239, "domain": "spatial_reasoning", "name": "SUNRGBD", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1086, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/sunrgbd?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-superchem", "caveat": "SuperChem is a benchmark of advanced chemistry problems requiring expert-level domain knowledge and reasoning.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SuperChem", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1087, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/superchem?top_n=500"}, {"aliases": [], "benchmark_id": "superchem", "caveat": "Multimodal chemical reasoning; image and formula parsing quality moves the reported figure.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "science", "name": "SUPERChem", "organization_count": 1, "organizations": ["Tencent"], "rank": 1088, "released": null, "source": "model_reports", "url": "https://github.com/catalystforyou/SUPERChem_eval"}, {"aliases": [], "benchmark_id": "llm-stats-superglue", "caveat": "SuperGLUE is a new benchmark styled after GLUE with a new set of more difficult language understanding tasks, improved resources, and a new public leaderboard. It includes 8 primary tasks: BoolQ (Boolean Questions), CB (CommitmentBank), COPA (Choice of Plausible Alternatives), MultiRC (Multi-Sentence Reading Comprehension), ReCoRD (Reading Comprehension with Commonsense Reasoning), RTE (Recognizing Textual Entailment), WiC (Word-in-Context), and WSC (Winograd Schema Challenge). The benchmark evaluates diverse language understanding capabilities including reading comprehension, commonsense reasoning, causal reasoning, coreference resolution, textual entailment, and word sense disambiguation across multiple domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SuperGLUE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1089, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/superglue?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-supergpqa", "caveat": "SuperGPQA is a comprehensive benchmark that evaluates large language models across 285 graduate-level academic disciplines. The benchmark contains 25,957 questions covering 13 broad disciplinary areas including Engineering, Medicine, Science, and Law, with specialized fields in light industry, agriculture, and service-oriented domains. It employs a Human-LLM collaborative filtering mechanism with over 80 expert annotators to create challenging questions that assess graduate-level knowledge and reasoning capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "SuperGPQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1090, "released": "2025-02-20", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/supergpqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1538-supergpqa", "caveat": "SuperGPQA, a comprehensive benchmark designed to evaluate the knowledge and reasoning abilities of Large Language Models (LLMs) across 285 graduate-level disciplines. SuperGPQA features at least 50 questions per discipline, covering a broad spectrum of graduate-level topics. SuperGPQA，这是一个旨在评估大型语言模型在 285 个研究生学科领域的知识和推理能力的全面基准。SuperGPQA 每个学科至少包含 50 个问题，涵盖广泛的硕士研究生学科主题。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SuperGPQA"], "document_share": 0.0008271298593879239, "domain": "学科", "name": "SuperGPQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1091, "released": "2025-02-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SuperGPQA"}, {"aliases": [], "benchmark_id": "llm-stats-surds", "caveat": "SURDS is a benchmark for spatial understanding and reasoning in autonomous-driving scenes.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "SURDS", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1092, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/surds?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1113-svamp", "caveat": "SVAMP includes one-unknown arithmetic word problems with grade level up to 4 by applying simple variations over word problems in an existing dataset. SVAMP further highlights the brittle nature of existing models when trained on these benchmark datasets. SVAMP 是一个包含算术文字问题的数据集，最高适用于四年级的学生，是通过对现有数据集中的文字问题应用简单变体而生成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SVAMP"], "document_share": 0.0008271298593879239, "domain": "数学", "name": "SVAMP", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1093, "released": "2021-04-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SVAMP"}, {"aliases": [], "benchmark_id": "llm-stats-svg-bench", "caveat": "SVG-Bench is an internal benchmark that comprehensively evaluates SVG generation performance. It accepts text and image inputs across build-from-scratch and edit-based tasks, using a VLM to verify rendering accuracy of the generated outputs.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "SVG-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1094, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/svg-bench?top_n=500"}, {"aliases": ["SWE Atlas", "SWE-Atlas"], "benchmark_id": "swe_atlas", "caveat": "Reports three splits (Codebase Q&A, Test Writing, Refactoring) as separate figures; the split is part of the instrument.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "coding_agent", "name": "SWE Atlas", "organization_count": 1, "organizations": ["Tencent"], "rank": 1095, "released": "2026-05-08", "source": "model_reports", "url": "https://github.com/scaleapi/SWE-Atlas"}, {"aliases": [], "benchmark_id": "llm-stats-swe-atlas-codebase-qna", "caveat": "SWE Atlas - Codebase QnA evaluates a model's ability to answer questions about real codebases, measuring repository-level comprehension and the ability to reason about code structure, behavior, and intent across an entire project.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "SWE Atlas - Codebase QnA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1096, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-codebase-qna?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-atlas-test-writing", "caveat": "SWE Atlas - Test Writing evaluates a model's ability to author meaningful tests for real-world software projects, measuring how well agents can understand code and produce correct, useful test coverage.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "SWE Atlas - Test Writing", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1097, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas-test-writing?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-atlas", "caveat": "SWE-Atlas is a software engineering benchmark focused on debugging, evaluating a model's ability to localize and fix bugs in real-world codebases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "SWE-Atlas", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1098, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-atlas?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-bench-multilingual", "caveat": "A multilingual benchmark for issue resolving in software engineering that covers Java, TypeScript, JavaScript, Go, Rust, C, and C++. Contains 1,632 high-quality instances carefully annotated from 2,456 candidates by 68 expert annotators, designed to evaluate Large Language Models across diverse software ecosystems beyond Python.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-bench Multilingual", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1099, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multilingual?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-bench-multimodal", "caveat": "SWE-Bench Multimodal extends SWE-Bench to evaluate language models on software engineering tasks that involve visual inputs such as screenshots, UI mockups, and diagrams alongside code understanding.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "SWE-Bench Multimodal", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1100, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-multimodal?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-bench-pro", "caveat": "SWE-Bench Pro is an advanced version of SWE-Bench that evaluates language models on complex, real-world software engineering tasks requiring extended reasoning and multi-step problem solving.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-Bench Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1101, "released": "2025-09-05", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-pro?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-bench-verified", "caveat": "A verified subset of 500 software engineering problems from real GitHub issues, validated by human annotators for evaluating language models' ability to resolve real-world coding issues by generating patches for Python codebases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-Bench Verified", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1102, "released": "2024-08-13", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-bench-verified-agentic-coding", "caveat": "SWE-bench Verified is a human-filtered subset of 500 software engineering problems drawn from real GitHub issues across 12 popular Python repositories. Given a codebase and an issue description, language models are tasked with generating patches that resolve the described problems. This benchmark evaluates AI's real-world agentic coding skills by requiring models to navigate complex codebases, understand software engineering problems, and coordinate changes across multiple functions, classes, and files to fix well-defined issues with clear descriptions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-bench Verified (Agentic Coding)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1103, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentic-coding%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-bench-verified-agentless", "caveat": "A human-validated subset of SWE-bench that evaluates language models' ability to resolve real-world GitHub issues using an agentless approach. The benchmark tests models on software engineering problems requiring understanding and coordinating changes across multiple functions, classes, and files simultaneously.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-bench Verified (Agentless)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1104, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28agentless%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-bench-verified-multiple-attempts", "caveat": "SWE-bench Verified is a human-validated subset of 500 test samples from the original SWE-bench dataset that evaluates AI systems' ability to automatically resolve real GitHub issues in Python repositories. Given a codebase and issue description, models must edit the code to successfully resolve the problem, requiring understanding and coordination of changes across multiple functions, classes, and files. The Verified version provides more reliable evaluation through manual validation of test samples.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-bench Verified (Multiple Attempts)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1105, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-bench-verified-%28multiple-attempts%29?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1984-swe-bench-live", "caveat": "SWE-bench-Live is a live-updatable benchmark designed for evaluating large language models (LLMs) and agents on real-world software issue resolution tasks. SWE-bench-Live 是一个面向大语言模型（LLMs）和智能体的实时可更新评测基准，专注于真实世界软件开发中的问题修复任务。 该基准从 2024 年以来的 GitHub 活跃仓库中自动收集了 1,319 个问题修复任务，涵盖 93 个项目，并为每个任务提供可复现的 Docker 执行环境。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SWE-bench-Live"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "SWE-bench-Live", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1106, "released": "2025-06-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SWE-bench-Live"}, {"aliases": [], "benchmark_id": "opencompass-1989-swe-factory", "caveat": "SWE-Factory is a benchmark for evaluating large language models on software issue fixing tasks. SWE-Factory 是一个面向大型语言模型的软件问题修复评测基准，旨在提升构建效率与评估准确性。该基准集成多智能体系统 SWE-Builder 自动搭建任务环境，采用退出码自动评分，并通过 fail2pass 流程验证修复有效性，确保评测可靠。SWE-Factory 覆盖四种语言共 671 个问题，支持高效、自动化、可扩展的 LLM 评估流程。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SWE-Factory"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "SWE-Factory", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1107, "released": "2025-06-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SWE-Factory"}, {"aliases": [], "benchmark_id": "llm-stats-swe-fficiency", "caveat": "SWE-fficiency is an open-source benchmark and workflow that evaluates language models on optimizing the runtime efficiency of real-world software engineering tasks, measuring how well agents can improve code performance autonomously.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "SWE-fficiency", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1108, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-fficiency?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-lancer", "caveat": "A benchmark for evaluating large language models on real-world freelance software engineering tasks from Upwork. Contains over 1,400 tasks valued at $1 million USD total, ranging from $50 bug fixes to $32,000 feature implementations. Includes both independent engineering tasks graded via end-to-end tests and managerial tasks assessed against original engineering managers' choices.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-Lancer", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1109, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-lancer-ic-diamond-subset", "caveat": "SWE-Lancer (IC-Diamond subset) is a benchmark of real-world freelance software engineering tasks from Upwork, ranging from $50 bug fixes to $32,000 feature implementations. It evaluates AI models on independent engineering tasks using end-to-end tests triple-verified by experienced software engineers, and includes managerial tasks where models choose between technical implementation proposals.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "SWE-Lancer (IC-Diamond subset)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1110, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-lancer-%28ic-diamond-subset%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-marathon", "caveat": "SWE-Marathon is an ultra-long-horizon software engineering benchmark covering tasks such as building compilers, optimizing kernels, and developing production-grade services. It measures whether agents can sustain quality across extremely long engineering trajectories.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "SWE-Marathon", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1111, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-marathon?top_n=500"}, {"aliases": ["SWE-Marathon", "SWE Marathon"], "benchmark_id": "swe_marathon", "caveat": "Ultra long-horizon software engineering tasks; the score depends on the agent's loop budget as much as on coding skill.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "coding_agent", "name": "SWE-Marathon", "organization_count": 1, "organizations": ["Tencent"], "rank": 1112, "released": null, "source": "model_reports", "url": "https://github.com/abundant-ai/swe-marathon"}, {"aliases": [], "benchmark_id": "llm-stats-swe-mm", "caveat": "SWE-MM evaluates software-engineering agents on repository tasks that require understanding both source code and visual evidence.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "SWE-MM", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1113, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-mm?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-perf", "caveat": "Software Engineering Performance benchmark measuring code optimization capabilities", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "SWE-Perf", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1114, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-perf?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-swe-review", "caveat": "Software Engineering Review benchmark evaluating code review capabilities", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "SWE-Review", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1115, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swe-review?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1608-swiltra-bench", "caveat": "SwiLTra-Bench is a comprehensive multilingual benchmark of over 180K aligned Swiss legal translation pairs comprising laws, headnotes, and press releases across all Swiss languages along with English, designed to evaluate LLM-based translation systems. SwiLTra-Bench是一个包含超过 18 万对对齐的瑞士法律翻译语料库的全面多语言基准，包括所有瑞士语言以及英语的法律、摘要和新闻稿，旨在评估基于LLM的翻译系统。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/SwiLTra-Bench"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "SwiLTra-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1116, "released": "2025-03-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/SwiLTra-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-swt-bench", "caveat": "Software Test Benchmark evaluating LLM ability to write tests for software repositories", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "SWT-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1117, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/swt-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-540-t-eval", "caveat": "T-Eval evaluates the tool utilization capabilities of LLMs and decomposing them into instruction following, planning, reasoning, retrieval, understanding, and review. T-Eval 评估了 LLM 的工具使用能力，并将其分解为指令遵循、规划、推理、检索、理解和审查等子能力", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/T-Eval"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "T-Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1118, "released": "2024-01-15", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/T-Eval"}, {"aliases": [], "benchmark_id": "llm-stats-t2-bench", "caveat": "t2-bench is a benchmark for evaluating agentic tool use capabilities, measuring how well models can select, sequence, and utilize tools to solve complex tasks. It tests autonomous planning and execution in multi-step scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "t2-bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1119, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/t2-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1132-tabfact", "caveat": "TabFac consists of 117,854 manually annotated statements with regard to 16,573 Wikipedia tables, their relations are classified as ENTAILED and REFUTED. TabFac 包含 117,854 条手动标注的语句，涉及 16,573 个维基百科表格，是第一个评估结构化数据上语言推理的数据集，涉及在符号和语言两个方面的混合推理能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TabFact"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "TabFact", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1120, "released": "2020-06-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TabFact"}, {"aliases": [], "benchmark_id": "opencompass-2025-tableeval", "caveat": "TableEval is the first cross-lingual benchmark for tabular question answering, supporting Simplified Chinese, Traditional Chinese, and English. It is designed to evaluate model performance on real-world, complex table understanding tasks across multiple languages. TableEval 是首个支持简体中文、繁体中文和英文的跨语言表格问答基准数据集，旨在系统评估大模型在真实复杂表格理解任务中的表现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TableEval"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "TableEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1121, "released": "2025-06-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TableEval"}, {"aliases": [], "benchmark_id": "opencompass-1164-taskbench", "caveat": "TaskBench aims to evaluate the capability of LLMs in task automation, containing 28,271 samples spanning 3 critical stages: task decomposition, tool invocation, and parameter prediction. TaskBench旨在评估LLM在任务自动化方面的能力，包含面向任务分解、工具调用和参数预测三个关键阶段的28271个样本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TaskBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "TaskBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1122, "released": "2023-11-30", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TaskBench"}, {"aliases": [], "benchmark_id": "llm-stats-tau-bench", "caveat": "τ-bench: A benchmark for tool-agent-user interaction in real-world domains. Tests language agents' ability to interact with users and follow domain-specific rules through dynamic conversations using API tools and policy guidelines across retail and airline domains. Evaluates consistency and reliability of agent behavior over multiple trials.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau-bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1123, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau-bench-airline", "caveat": "Part of τ-bench (TAU-bench), a benchmark for Tool-Agent-User interaction in real-world domains. The airline domain evaluates language agents' ability to interact with users through dynamic conversations while following domain-specific rules and using API tools. Agents must handle airline-related tasks and policies reliably.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "TAU-bench Airline", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1124, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-airline?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau-bench-retail", "caveat": "A benchmark for evaluating tool-agent-user interaction in retail environments. Tests language agents' ability to handle dynamic conversations with users while using domain-specific API tools and following policy guidelines. Evaluates agents on tasks like order cancellations, address changes, and order status checks through multi-turn conversations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "TAU-bench Retail", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1125, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau-bench-retail?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau2-airline", "caveat": "TAU2 airline domain benchmark for evaluating conversational agents in dual-control environments where both AI agents and users interact with tools in airline customer service scenarios. Tests agent coordination, communication, and ability to guide user actions in tasks like flight booking, modifications, cancellations, and refunds.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau2 Airline", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1126, "released": "2025-06-09", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-airline?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau2-retail", "caveat": "τ²-bench retail domain evaluates conversational AI agents in customer service scenarios within a dual-control environment where both agent and user can interact with tools. Tests tool-agent-user interaction, rule adherence, and task consistency in retail customer support contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau2 Retail", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1127, "released": "2025-06-09", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-retail?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau2-telecom", "caveat": "τ²-Bench telecom domain evaluates conversational agents in a dual-control environment modeled as a Dec-POMDP, where both agent and user use tools in shared telecommunications troubleshooting scenarios that test coordination and communication capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau2 Telecom", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1128, "released": "2025-06-09", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau2-telecom?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau3-airline", "caveat": "τ³-Bench airline domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated airline booking and reservations environment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau3 Airline", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1129, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-airline?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau3-banking", "caveat": "τ³-Bench banking domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated retail banking environment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau3 Banking", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1130, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-banking?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau3-retail", "caveat": "τ³-Bench retail domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated online retail environment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau3 Retail", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1131, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-retail?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau3-telecom", "caveat": "τ³-Bench telecom domain evaluates agentic models on multi-turn, tool-using customer-support and troubleshooting scenarios in a simulated telecommunications environment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Tau3 Telecom", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1132, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-telecom?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tau3-bench", "caveat": "TAU3-Bench is a benchmark for evaluating general-purpose agent capabilities, testing models on multi-turn interactions with simulated user models, retrieval, and complex decision-making scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "TAU3-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1133, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tau3-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tempcompass", "caveat": "TempCompass is a comprehensive benchmark for evaluating temporal perception capabilities of Video Large Language Models (Video LLMs). It constructs conflicting videos that share identical static content but differ in specific temporal aspects to prevent models from exploiting single-frame bias. The benchmark evaluates multiple temporal aspects including action, motion, speed, temporal order, and attribute changes across diverse task formats including multi-choice QA, yes/no QA, caption matching, and caption generation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "TempCompass", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1134, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tempcompass?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-terminal-bench", "caveat": "Terminal-Bench is a benchmark for testing AI agents in real terminal environments. It evaluates how well agents can handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, security tasks, data science workflows, and cybersecurity vulnerabilities. The benchmark consists of a dataset of ~100 hand-crafted, human-verified tasks and an execution harness that connects language models to a terminal sandbox.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Terminal-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1135, "released": "2025-01-17", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-terminal-bench-2", "caveat": "Terminal-Bench 2.0 is an updated benchmark for testing AI agents' tool use ability to operate a computer via terminal. It evaluates how well models can handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, security tasks, data science workflows, and cybersecurity vulnerabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Terminal-Bench 2.0", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1136, "released": "2025-09-25", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-terminal-bench-2-1", "caveat": "Terminal-Bench 2.1 is an updated release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal. It evaluates how well models handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, data science workflows, and security tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Terminal-Bench 2.1", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1137, "released": "2026-05-05", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-2.1?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-terminal-bench-3-0", "caveat": "Terminal-Bench 3.0 is a release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal on real-world, end-to-end tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Terminal-Bench 3.0", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1138, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-3.0?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-terminal-bench-hard", "caveat": "Agentic coding & terminal use", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/terminalbench-hard"], "document_share": 0.0008271298593879239, "domain": "coding", "name": "Terminal-Bench Hard", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 1139, "released": "2025-09-02", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/terminalbench-hard"}, {"aliases": [], "benchmark_id": "llm-stats-terminal-bench-hard", "caveat": "Terminal-Bench Hard is a harder terminal-agent benchmark variant evaluated with the Terminus-2 harness in Cohere's Command A+ and North Mini Code releases.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Terminal-Bench Hard", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1140, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminal-bench-hard?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-terminal-bench-v2-1", "caveat": "Agentic coding & terminal use", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/terminalbench-v2-1"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "Terminal-Bench v2.1", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 1141, "released": "2026-05-06", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1"}, {"aliases": [], "benchmark_id": "llm-stats-terminus", "caveat": "Terminal-Bench is a benchmark for testing AI agents in real terminal environments, evaluating how well agents can handle real-world, end-to-end tasks autonomously. The benchmark includes tasks spanning coding, system administration, security, data science, model training, file operations, version control, and web development. Terminus is the neutral test-bed agent designed to work with Terminal-Bench, operating purely through tmux sessions without dedicated tools.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Terminus", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1142, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/terminus?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-community-fd462fc2-283c-4967-bd7d-b39d7c661807", "caveat": null, "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/community%3Afd462fc2-283c-4967-bd7d-b39d7c661807?top_n=500"], "document_share": 0.0008271298593879239, "domain": "other", "name": "testing", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1143, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/community%3Afd462fc2-283c-4967-bd7d-b39d7c661807?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1574-text2world", "caveat": "Text2World, based on planning domain definition language (PDDL), featuring hundreds of diverse domains and employing multi-criteria, execution-based metrics for a more robust evaluation. Text2World 基于规划领域定义语言 (PDDL)，拥有数百个不同的领域，并采用多标准、基于执行的衡量标准来进行更稳健的评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Text2World"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "Text2World", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1144, "released": "2025-02-18", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Text2World"}, {"aliases": [], "benchmark_id": "llm-stats-textvqa", "caveat": "TextVQA contains 45,336 questions on 28,408 images that require reasoning about text to answer. Introduced to benchmark VQA models' ability to read and reason about text within images, particularly for assistive technologies for visually impaired users. The dataset addresses the gap where existing VQA datasets had few text-based questions or were too small.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "TextVQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1145, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/textvqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1846-tglg", "caveat": "Temporally-Grounded Language Generation (TGLG) is a benchmark for real-time vision-language models (VLMs) that focus on two key capabilities: perceptual updating and contingency awareness. 基于时间的语言生成（TGLG）是实时视觉语言模型（VLM）的基准，侧重于两个关键功能：感知更新和应急意识。该存储库还包含TGLG的基线实时VLM代码，即具有时间同步交织的视觉语言模型（VLM-TSI）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TGLG"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "TGLG", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1146, "released": "2025-05-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TGLG"}, {"aliases": [], "benchmark_id": "opencompass-1734-thai-local-benchmark", "caveat": "Thai local dialect benchmark covers Northern (Lanna), Northeastern (Isan), and Southern (Dambro) Thai, evaluating LLMs on five NLP tasks: summarization, question answering, translation, conversation, and food-related tasks. 这是一个涵盖泰国北部（兰纳）、东北部（伊森）和南部（丹布罗）方言的泰国地方方言基准测试，评估大型语言模型在五项自然语言处理任务上的表现：总结、问答、翻译、对话以及与食物相关的任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Thai_local_benchmark"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "Thai_local_benchmark", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1147, "released": "2025-04-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Thai_local_benchmark"}, {"aliases": [], "benchmark_id": "llm-stats-theoremqa", "caveat": "A theorem-driven question answering dataset containing 800 high-quality questions covering 350+ theorems from Math, Physics, EE&CS, and Finance. Designed to evaluate AI models' capabilities to apply theorems to solve challenging university-level science problems.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "TheoremQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1148, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/theoremqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1134-theoremqa", "caveat": "TheoremQA is the first theorem-driven question-answering dataset designed to evaluate AI models’ capabilities to apply theorems to solve challenging science problems. It is curated by domain experts containing 800 high-quality questions covering 350 theorems from Math, Physics, EE&CS, and Finance. TheoremQA 是第一个基于定理的问题回答数据集，旨在评估 AI 模型应用定理解决复杂科学问题的能力。该数据集由领域专家精心策划，包含 800 个高质量问题，涵盖来自数学、物理、电气与计算机科学以及金融的 350 个定理。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TheoremQA"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "TheoremQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1149, "released": "2023-12-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TheoremQA"}, {"aliases": [], "benchmark_id": "opencompass-2391-threat-signature-eval", "caveat": "Threat Signature Eval is a domain-specific benchmark for evaluating video-language models on real-world physical security / surveillance footage. It measures how well a model can recognize and classify security-relevant events into a 10-category threat taxonomy (e.g., Fire & Smoke, Fighting & Violen Threat Signature Eval 是一个面向物理安防场景的领域基准，用于评测视频语言模型在真实监控摄像头画面中的理解与事件识别能力。该基准要求模型在安全相关的视频片段中识别并分类事件（10 类威胁签名，如火灾烟雾、斗殴暴力、非法入侵等），相关介绍与结果在 Ambient.ai 的 Pulsar VLM 发布主题演讲中给出。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Threat-Signature_Eval"], "document_share": 0.0008271298593879239, "domain": "视觉定位", "name": "Threat-Signature_Eval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1150, "released": "2025-11-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Threat-Signature_Eval"}, {"aliases": [], "benchmark_id": "opencompass-2050-thunder", "caveat": "THUNDER is a benchmark designed to evaluate digital pathology foundation models on tile-level image understanding tasks, facilitating comparative analysis across various downstream tasks and models. THUNDER 是一个用于评估数字病理学基础模型在切片级图像理解任务中表现的基准，旨在支持多种下游任务和模型的对比分析。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/THUNDER"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "THUNDER", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1151, "released": "2025-07-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/THUNDER"}, {"aliases": [], "benchmark_id": "opencompass-1668-timetravel", "caveat": "TimeTravel Taxonomy maps artifacts from 10 civilizations, 266 cultures, and 10k+ verified samples for AI-driven historical analysis. 时间旅行分类将来自 10 个文明、266 个文化以及 10k+个验证样本的文物映射，用于 AI 驱动的历史分析。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TimeTravel"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "TimeTravel", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1152, "released": "2025-02-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TimeTravel"}, {"aliases": [], "benchmark_id": "opencompass-1839-tiny-qa-benchmark-pp", "caveat": "TinyQA is a benchmark suite designed to evaluate the reasoning abilities of large language models (LLMs). It focuses on assessing LLMs through natural language question-answer pairs, covering various types of reasoning tasks such as causal, logical, and commonsense reasoning. TinyQA是一个用于评估大语言模型（LLMs）推理能力的基准测试套件。该基准专注于通过自然语言问题和答案对来衡量LLMs的推理能力，涵盖了多种类型的推理任务，包括因果推理、逻辑推理和常识推理。TinyQA提供了多样化的数据集，旨在挑战LLMs的推理深度和广度。通过严格的评估，TinyQA能够帮助研究人员更好地理解LLMs在处理复杂语言任务时的表现，并为改进模型提供方向。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/tiny_qa_benchmark_pp"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "tiny_qa_benchmark_pp", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1153, "released": "2025-05-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/tiny_qa_benchmark_pp"}, {"aliases": [], "benchmark_id": "llm-stats-tir-bench", "caveat": "A tool-calling and multimodal interaction benchmark for testing visual instruction following and execution reliability.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "TIR-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1154, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tir-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tldr9-test", "caveat": "A large-scale summarization dataset containing over 9 million training instances extracted from Reddit, designed for extreme summarization (generating one-sentence summaries with high compression and abstraction). More than twice larger than previously proposed datasets.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "summarization", "name": "TLDR9+ (test)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1155, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tldr9%2B-%28test%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tomato", "caveat": "TOMATO (Temporal Reasoning Multimodal Evaluation) assesses multimodal models on motion and temporal perception in video, testing understanding of actions, motion, and changes over time.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "TOMATO", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1156, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tomato?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-toolathlon", "caveat": "Tool Decathlon is a comprehensive benchmark for evaluating AI agents' ability to use multiple tools across diverse task categories. It measures proficiency in tool selection, sequencing, and execution across ten different tool-use scenarios.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Toolathlon", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1157, "released": "2025-10-29", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/toolathlon?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1712-toolhop", "caveat": "ToolHop is a dataset specifically designed for rigorous evaluation of multi-hop tool use, which ensures diverse queries, meaningful interdependencies, locally executable tools, detailed feedback, and verifiable answers through a novel query-driven data construction approach. ToolHop是一个通过查询驱动构建的数据集，专门用于评测大模型的多跳工具使用能力，具备多样化的查询、有意义的相互依赖关系、本地可执行的工具、详细的反馈和可验证的答案五大特征。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ToolHop"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "ToolHop", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1158, "released": "2025-01-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ToolHop"}, {"aliases": [], "benchmark_id": "opencompass-1607-toolret", "caveat": "ToolRet is a heterogeneous tool retrieval benchmark comprising 7.6k diverse retrieval tasks, and a corpus of 43k tools, collected from existing datasets. ToolRet是一个包含 7.6k 个不同检索任务的异构工具检索基准，以及从现有数据集中收集的 43k 个工具语料库。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ToolRet"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "ToolRet", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1159, "released": "2025-03-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ToolRet"}, {"aliases": [], "benchmark_id": "llm-stats-trae-code-gen", "caveat": "Trae Code Gen is a component of Trae Agent Bench that evaluates implementing new functionality across multiple programming languages in containerized, runnable repositories.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Trae Code Gen", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1160, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-code-gen?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-trae-error-fix", "caveat": "Trae Error Fix is a component of Trae Agent Bench that evaluates fixing existing code across multiple programming languages in containerized, runnable repositories.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Trae Error Fix", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1161, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/trae-error-fix?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1855-transbench", "caveat": "TransBench is the first industry-oriented comprehensive multilingual translation evaluation system designed for industrial applications. It quantifies translation model performance across diverse industries and linguistic environments through meticulously curated datasets aligned with standards. TransBench is the first industry-oriented comprehensive multilingual translation evaluation system designed for industrial applications. It quantifies translation model performance across diverse industries and linguistic environments through meticulously curated datasets aligned with standards.", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TransBench"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "TransBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1162, "released": "2025-05-20", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TransBench"}, {"aliases": [], "benchmark_id": "llm-stats-translation-en-set1-comet22", "caveat": "COMET-22 is an ensemble machine translation evaluation metric combining a COMET estimator model trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It demonstrates improved correlations compared to state-of-the-art metrics and increased robustness to critical errors.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "Translation en→Set1 COMET22", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1163, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-comet22?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-translation-en-set1-spbleu", "caveat": "Translation evaluation using spBLEU (SentencePiece BLEU), a BLEU metric computed over text tokenized with a language-agnostic SentencePiece subword model. Introduced in the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "Translation en→Set1 spBleu", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1164, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-en%E2%86%92set1-spbleu?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-translation-set1-en-comet22", "caveat": "COMET-22 is a neural machine translation evaluation metric that uses an ensemble of two models: a COMET estimator trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It provides improved correlations with human judgments and increased robustness to critical errors compared to previous metrics.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "Translation Set1→en COMET22", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1165, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-comet22?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-translation-set1-en-spbleu", "caveat": "spBLEU (SentencePiece BLEU) evaluation metric for machine translation quality assessment, using language-agnostic SentencePiece tokenization with BLEU scoring. Part of the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "Translation Set1→en spBleu", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1166, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/translation-set1%E2%86%92en-spbleu?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2039-translaw", "caveat": "本文提出TransLaw——专为香港判例翻译设计的协同交互式多智能体框架。该框架创新性地将传统翻译流程拆解为翻译、错误标注及校对修正三大子任务，并分配三个智能体协同执行。为评估框架性能，我们构建了大规模双语基准数据集BJC Judgments，对13个开源与商业大语言模型（作为智能体）展开评测。实验结果验证了协同策略的有效性：在多智能体协作显著提升效果的同时，提供了具有参考价值的LLM性能横向对比。通过错误类型学分析，本研究进一步揭示了亟待解决的关键翻译挑战。未来工作将聚焦于优化智能体架构以应对这些挑战，同时开发更全面、低成本的评估基准。 本文提出TransLaw——专为香港判例翻译设计的协同交互式多智能体框架。该框架创新性地将传统翻译流程拆解为翻译、错误标注及校对修正三大子任务，并分配三个智能体协同执行。为评估框架性能，我们构建了大规模双语基准数据集BJC Judgments，对13个开源与商业大语言模型（作为智能体）展开评测。实验结果验证了协同策略的有效性：在多智能体协作显著提升效果的同时，提供了具有参考价值的LLM性能横向对比。通过错误类型学分析，本研究进一步揭示了亟待解决的关键翻译挑战。未来工作将聚焦于优化智能体架构以应对这些挑战，同时开发更全面、低成本的评估基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TransLaw"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "TransLaw", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1167, "released": "2025-07-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TransLaw"}, {"aliases": [], "benchmark_id": "llm-stats-treebench", "caveat": "TreeBench evaluates visual grounded reasoning, requiring models to localize and reason about fine-grained visual details.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "TreeBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1168, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/treebench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-triviaqa", "caveat": "A large-scale reading comprehension dataset containing over 650K question-answer-evidence triples. TriviaQA includes 95K question-answer pairs authored by trivia enthusiasts and independently gathered evidence documents (six per question on average) that provide high quality distant supervision for answering the questions. The dataset features relatively complex, compositional questions with considerable syntactic and lexical variability, requiring cross-sentence reasoning to find answers.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "TriviaQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1169, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/triviaqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-512-triviaqa", "caveat": "TriviaqQA is a reading comprehension dataset containing over 650K question-answer-evidence triples. TriviaqQA includes 95K question-answer pairs authored by trivia enthusiasts and independently gathered evidence documents, six per question on average, that provide high quality distant supervision for answering the questions. TriviaqQA是一个阅读理解数据集，包含超过65万个问题-答案-证据三元组。其包括95K个问答对，由冷知识爱好者和独立收集的事实性文档撰写，平均每个问题6个，为回答问题提供高质量的远程监督。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TriviaQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "TriviaQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1170, "released": "2017-05-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TriviaQA"}, {"aliases": [], "benchmark_id": "llm-stats-truthfulqa", "caveat": "TruthfulQA is a benchmark to measure whether language models are truthful in generating answers to questions. It comprises 817 questions that span 38 categories, including health, law, finance and politics. The questions are crafted such that some humans would answer falsely due to a false belief or misconception, testing models' ability to avoid generating false answers learned from human texts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "TruthfulQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1171, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/truthfulqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1096-truthfulqa", "caveat": "TruthfulQA is  a benchmark to measure whether a language model is truthful in generating answers to questions. The benchmark comprises 817 questions that span 38 categories, including health, law, finance and politics. TruthfulQA 用于测量语言模型在回答问题时的真实度。该基准包含 817 个问题，涵盖 38 个类别，包括健康、法律、金融和政治。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TruthfulQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "TruthfulQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1172, "released": "2022-05-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TruthfulQA"}, {"aliases": [], "benchmark_id": "llm-stats-tvbench", "caveat": "TVBench is a temporal video understanding benchmark evaluating reasoning over actions, events, and temporal dynamics in videos.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "TVBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1173, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tvbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-tydiqa", "caveat": "A multilingual question answering benchmark covering 11 typologically diverse languages with 204K question-answer pairs. Questions are written by people seeking genuine information and data is collected directly in each language without translation to test model generalization across diverse linguistic structures.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "TydiQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1174, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/tydiqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-508-tydiqa", "caveat": "TyDi QA is a question answering dataset covering 11 typologically diverse languages with 204K question-answer pairs. The languages of TyDi QA are diverse with regard to their typology -- the set of linguistic features that each language expresses. TyDi QA 是一个涵盖 11 种不同语言的问题回答数据集，包含 20.4 万个问题-答案对。TyDi QA 的语言种类多样，涵盖了语言学特征的各种类型。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/TyDiQA"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "TyDiQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1175, "released": "2020-03-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/TyDiQA"}, {"aliases": [], "benchmark_id": "opencompass-1742-u-niah", "caveat": "U-NIAH is a framework unifying RAG and LLM for needle-in-a-haystack tasks, based on the fictional Starlight Academy dataset. It eliminates interference from pre-trained knowledge and supports diverse, complex scenarios (e.g., multi-needle, long-needle, \"needle-in-needle\"). U-NIAH是将 RAG 和 LLM 统一映射在大海捞针任务中的框架。所有任务基于一个虚构背景下的数据集Starlight Academy，涵盖了魔法系统、学术课程、校园生活、等多个方面，旨在消除预训练知识的干扰，从而能够独立于 LLMs 的先验知识。框架包含多种评估场景，支持多针（3、7、15个针）和长针（400-500 token）配置，还引入了“针中针”结构，进一步增加了复杂性。该数据集通过多样化的场景和合成生成的内容，能够从多个维度分析模型在长文本场景下的性能。同时通过模块化设计，U-NIAH可持续注入新的挑战场景。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/U-NIAH"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "U-NIAH", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1176, "released": "2025-03-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/U-NIAH"}, {"aliases": [], "benchmark_id": "opencompass-1625-ubuntu-osworld", "caveat": "OSWorld is a first-of-its-kind scalable, real computer environment for multimodal agents, supporting task setup, execution-based evaluation, and interactive learning across operating systems. OSWorld 是一个首创的、可扩展的、真实计算机环境，用于多模态智能体，支持操作系统跨平台的任务设置、基于执行的评估和交互式学习。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ubuntu_osworld"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ubuntu_osworld", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1177, "released": "2024-04-11", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ubuntu_osworld"}, {"aliases": [], "benchmark_id": "opencompass-1088-uhgeval", "caveat": "UHGEval(Unconstrained Hallucination Generation Evaluation) benchmark contains hallucinations generated by LLMs with minimal restrictions. UHGEval 基准，包含由限制条件最小的大语言模型（LLMs）生成的幻觉。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UHGEval"], "document_share": 0.0008271298593879239, "domain": "知识", "name": "UHGEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1178, "released": "2024-05-24", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/UHGEval"}, {"aliases": [], "benchmark_id": "opencompass-1332-unibench", "caveat": "UniBench is meant for evaluating VLMs' reasoning abilities. It is a unified implementation of 50+ VLM benchmarks spanning a comprehensive range of carefully categorized capabilities from object recognition to spatial awareness, counting, and much more. UniBench旨在评估VLM的推理能力，包括50个基准测试，涵盖对象识别、空间感知、计数等任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UniBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "UniBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1179, "released": "2024-08-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/UniBench"}, {"aliases": [], "benchmark_id": "llm-stats-uniform-bar-exam", "caveat": "The Uniform Bar Examination (UBE) benchmark evaluates language models on the complete bar exam including multiple-choice Multistate Bar Examination (MBE), open-ended Multistate Essay Exam (MEE), and Multistate Performance Test (MPT) components. Used to assess legal reasoning capabilities across seven subject areas including Evidence, Torts, Constitutional Law, Contracts, Criminal Law and Procedure, Real Property, and Civil Procedure.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "Uniform Bar Exam", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1180, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/uniform-bar-exam?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1636-urbanvideo-bench", "caveat": "The benchmark is designed to evaluate whether video-large language models (Video-LLMs) can naturally process continuous first-person visual observations like humans, enabling recall, perception, reasoning, and navigation. UrbanVideo-Bench旨在评估视频大型语言模型（Video-LLMs）是否能够像人类一样自然地处理连续的第一人称视觉观察，实现回忆、感知、推理和导航。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UrbanVideo-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "UrbanVideo-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1181, "released": "2025-03-08", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/UrbanVideo-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-usamo-2026", "caveat": "USAMO 2026 evaluates models on the six problems from the 2026 United States of America Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "USAMO 2026", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1182, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo-2026?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-usamo25", "caveat": "The 2025 United States of America Mathematical Olympiad (USAMO) benchmark consists of six challenging mathematical problems requiring rigorous proof-based reasoning. USAMO is the most prestigious high school mathematics competition in the United States, serving as the final round of the American Mathematics Competitions series. This benchmark evaluates models on mathematical problem-solving capabilities beyond simple numerical computation, focusing on formal mathematical reasoning and proof generation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "USAMO25", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1183, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/usamo25?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2014-utboost", "caveat": "UTBoost generates unit tests using LLMs to augment the test cases for certain instances in SWE-Bench, enabling a more rigorous use of SWE-Bench to evaluate the performance of Code Agents. UTBoost通过LLM生成的单元测试，增强了SWE-Bench中一些instances的测试用例，能严谨的使用SWE-Bench来评估Code Agents的表现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/UTBoost"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "UTBoost", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1184, "released": "2025-06-10", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/UTBoost"}, {"aliases": [], "benchmark_id": "llm-stats-v-star", "caveat": "A visual reasoning benchmark evaluating multimodal inference under challenging spatial and grounded tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "V*", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1185, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/v-star?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1654-v-star", "caveat": "V-STaR is a spatio-temporal reasoning benchmark for Video-LLMs, evaluating Video-LLM’s spatio-temporal reasoning ability in answering questions explicitly in the context of “when”, “where”, and “what”. V-STaR 是一个针对 Video-LLMs的空间时间推理基准，评估 Video-LLM在“何时”、“何地”和“何物”的上下文中明确回答问题的空间时间推理能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/V-STaR"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "V-STaR", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1186, "released": "2025-03-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/V-STaR"}, {"aliases": [], "benchmark_id": "llm-stats-vatex", "caveat": "VaTeX: A Large-Scale, High-Quality Multilingual Dataset for Video-and-Language Research. Contains over 41,250 videos and 825,000 captions in both English and Chinese, with over 206,000 English-Chinese parallel translation pairs. Supports multilingual video captioning and video-guided machine translation tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VATEX", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1187, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vatex?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1375-vbench", "caveat": "VBench is a comprehensive benchmark evaluates video generation quality. It comprises 16 dimensions in video generation, and also provides a dataset of human preference annotations. VBench用于评估多模态大模型的视频生成质量，包含16个视频生成维度及1个人类偏好注释数据集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "VBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1188, "released": "2023-11-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VBench"}, {"aliases": [], "benchmark_id": "llm-stats-vcr-en-easy", "caveat": "Visual Commonsense Reasoning (VCR) benchmark that tests higher-order cognition and commonsense reasoning beyond simple object recognition. Models must answer challenging questions about images and provide rationales justifying their answers. The benchmark measures the ability to infer people's actions, goals, and mental states from visual context.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "VCR_en_easy", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1189, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vcr-en-easy?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vending-bench-2", "caveat": "Vending-Bench 2 tests longer horizon planning capabilities by evaluating how well AI models can manage a simulated vending machine business over extended periods. The benchmark measures a model's ability to maintain consistent tool usage and decision-making for a full simulated year of operation, driving higher returns without drifting off task.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Vending-Bench 2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1190, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vending-bench-2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe", "caveat": "Visual Interface Building Evaluation benchmark for UI/app generation", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "VIBE", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1191, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1875-vibe", "caveat": "SVRPBench is an open and extensible benchmark for the Stochastic Vehicle Routing Problem (SVRP). It includes 500+ instances spanning small to large scales (10–1000 customers), designed to evaluate algorithms under realistic urban logistics conditions with uncertainty and operational constraints. SVRPBench是一个针对随机车辆路径问题（SVRP）的开放且可扩展的基准测试平台。它包含500多个实例，涵盖小到大规模（10-1000个客户），旨在评估算法在具有不确定性和操作约束的现实城市物流条件下的表现。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VIBE"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "VIBE", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1192, "released": "2025-05-29", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VIBE"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-android", "caveat": "VIBE benchmark subset for Android application generation", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "VIBE Android", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1193, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-android?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-backend", "caveat": "VIBE benchmark subset for backend service generation", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "VIBE Backend", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1194, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-backend?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-ios", "caveat": "VIBE benchmark subset for iOS application generation", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "VIBE iOS", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1195, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-ios?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-simulation", "caveat": "VIBE benchmark subset for simulation code generation", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "VIBE Simulation", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1196, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-simulation?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-web", "caveat": "VIBE benchmark subset for web application generation", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500"], "document_share": 0.0008271298593879239, "domain": "code", "name": "VIBE Web", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1197, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-web?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-eval", "caveat": "VIBE-Eval is a hard evaluation suite for measuring progress of multimodal language models, consisting of 269 visual understanding prompts with gold-standard responses authored by experts. The benchmark has dual objectives: vibe checking multimodal chat models for day-to-day tasks and rigorously testing frontier models, with the hard set containing >50% questions that all frontier models answer incorrectly.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Vibe-Eval", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1198, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-eval?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-pro", "caveat": "VIBE-Pro is an advanced version of the VIBE (Visual & Interactive Benchmark for Execution) benchmark that evaluates LLMs on professional-grade full-stack application development tasks. It measures model performance across complex real-world development scenarios including web, mobile, and backend applications with higher difficulty than the standard VIBE benchmark.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "VIBE-Pro", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1199, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-pro?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vibe-v2", "caveat": "VIBE-V2 is an internal benchmark covering pure front-end and full-stack Web, Android, and iOS projects with build-from-scratch tasks. It uses an Agent-as-a-Verifier paradigm to automatically verify program interaction logic and visual output, scoring models through a unified pipeline that includes a requirement set, containerized deployment, and a dynamic interaction environment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "VIBE-V2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1200, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vibe-v2?top_n=500"}, {"aliases": ["ViBench"], "benchmark_id": "vibench", "caveat": "End-to-end vibe-coding benchmark by authors at Replit and Georgian AI Lab, scoring web applications from the user's perspective. Tasks are derived from anonymized Replit production traces, so the benchmark is first-party to one of the agents it measures. It reaches this registry through a customer testimonial in Anthropic's Claude Fable 5 / Mythos 5 launch post, where it is described qualitatively and never scored in the comparison table. Distinct from VBench, an unrelated video-generation benchmark.", "document_count": 1, "document_ids": ["model_reports:anthropic_claude_fable_5_mythos_5"], "document_share": 0.0008271298593879239, "domain": "coding_agent", "name": "ViBench", "organization_count": 1, "organizations": ["Anthropic"], "rank": 1201, "released": "2026-05-26", "source": "model_reports", "url": "https://vibench.ai/"}, {"aliases": [], "benchmark_id": "llm-stats-video-mme", "caveat": "Video-MME is the first-ever comprehensive evaluation benchmark of Multi-modal Large Language Models (MLLMs) in video analysis. It features 900 videos totaling 254 hours with 2,700 human-annotated question-answer pairs across 6 primary visual domains (Knowledge, Film & Television, Sports Competition, Life Record, Multilingual, and others) and 30 subfields. The benchmark evaluates models across diverse temporal dimensions (11 seconds to 1 hour), integrates multi-modal inputs including video frames, subtitles, and audio, and uses rigorous manual labeling by expert annotators for precise assessment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Video-MME", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1202, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1358-video-mme", "caveat": "Video-MME is an evaluation benchmark of multi-modal LLMs in video analysis, including 900 videos in various duration with a total of 254 hours which spans 6 primary visual domains with 30 subfields. Video-MME用于评估多模态大模型的视频分析能力，包含900个不同长度的视频，来自6个主要视觉领域和30个子领域，总时长达254小时。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Video-MME"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "Video-MME", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1203, "released": "2024-03-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Video-MME"}, {"aliases": [], "benchmark_id": "llm-stats-video-mme-long-no-subtitles", "caveat": "Video-MME is the first-ever comprehensive evaluation benchmark for Multi-modal Large Language Models (MLLMs) in video analysis. This variant focuses on long-term videos (30min-60min) without subtitle inputs, testing robust contextual dynamics across 6 primary visual domains with 30 subfields including knowledge, film & television, sports competition, life record, and multilingual content.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Video-MME (long, no subtitles)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1204, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/video-mme-%28long%2C-no-subtitles%29?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1277-videogui", "caveat": "VideoGUI is designed to evaluate GUI assistants on visual-centric GUI tasks. Sourced from high-quality web instructional videos, it focuses on tasks involving professional and novel software and complex activities (e.g., video editing). VideoGUI旨在评估以视觉为中心的GUI任务上的GUI助手，来自高质量的网络教学视频，侧重于涉及专业和新颖软件和复杂活动（例如视频编辑）的任务。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VideoGUI"], "document_share": 0.0008271298593879239, "domain": "创作", "name": "VideoGUI", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1205, "released": "2024-06-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VideoGUI"}, {"aliases": [], "benchmark_id": "llm-stats-videoholmes", "caveat": "VideoHolmes evaluates video understanding and reasoning capabilities in multimodal models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VideoHolmes", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1206, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/videoholmes?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1926-videomathqa", "caveat": "VideoMathQA is a benchmark designed to evaluate mathematical reasoning in real-world educational videos. It requires models to interpret and integrate information from three modalities, visuals, audio, and text, across time. VideoMathQA是一个旨在评估实际教育视频中数学推理能力的基准。它要求模型解释和整合来自三种模态(视觉、音频和文本)随时间变化的信息。该基准解决了\"多模态针堆\"问题,即关键信息稀疏且分散在视频的不同模态和时刻。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VideoMathQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "VideoMathQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1207, "released": "2025-06-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VideoMathQA"}, {"aliases": [], "benchmark_id": "llm-stats-videomme-w-sub", "caveat": "The first-ever comprehensive evaluation benchmark of Multi-modal LLMs in Video analysis. Features 900 videos (254 hours) with 2,700 question-answer pairs covering 6 primary visual domains and 30 subfields. Evaluates temporal understanding across short (11 seconds) to long (1 hour) videos with multi-modal inputs including video frames, subtitles, and audio.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VideoMME w sub.", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1208, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-sub.?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-videomme-w-o-sub", "caveat": "Video-MME is a comprehensive evaluation benchmark for multi-modal large language models in video analysis. It features 900 videos across 6 primary visual domains with 30 subfields, ranging from 11 seconds to 1 hour in duration, with 2,700 question-answer pairs. The benchmark evaluates MLLMs' capabilities in processing sequential visual data and multi-modal content including video frames, subtitles, and audio.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VideoMME w/o sub.", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1209, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/videomme-w-o-sub.?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-videommmu", "caveat": "Video-MMMU evaluates Large Multimodal Models' ability to acquire knowledge from expert-level professional videos across six disciplines through three cognitive stages: perception, comprehension, and adaptation. Contains 300 videos and 900 human-annotated questions spanning Art, Business, Science, Medicine, Humanities, and Engineering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VideoMMMU", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1210, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/videommmu?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1900-videoreasonbench", "caveat": "VideoReasonBench is designed to evaluate vision-centric complex video reasoning. VideoReasonBench是一个用于评测视觉为中心、复杂视频推理的基准。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VideoReasonBench"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "VideoReasonBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1211, "released": "2025-06-05", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VideoReasonBench"}, {"aliases": [], "benchmark_id": "llm-stats-videosimpleqa", "caveat": "VideoSimpleQA evaluates factual knowledge grounded in video content, measuring how accurately models answer short, fact-seeking questions about videos.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VideoSimpleQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1212, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/videosimpleqa?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1717-vilbench", "caveat": "ViLBench is a benchmark designed to evaluate vision-language models. It features 600 examples from 5 datasets, selected based on the criterion that process reward models offer greater improvements over output reward models in guiding generations. ViLBench 是一项旨在评估视觉-语言模型的数据集，其强调对模型进行细粒度的逐步推理能力测试。该基准共包含600个经过严格筛选的样本，来源于五个不同的视觉-语言数据集，筛选标准是在模型答案选择过程中，过程奖励模型（process-reward model）相较于输出奖励模型（output-reward model）具有更显著的性能提升效果。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ViLBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ViLBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1213, "released": "2025-03-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ViLBench"}, {"aliases": [], "benchmark_id": "llm-stats-vct", "caveat": "Virology Capabilities Test (VCT) is an expert-level multiple-choice benchmark measuring the capability to troubleshoot complex virology laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "Virology Capabilities Test", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1214, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vct?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2063-visco", "caveat": "VISCO aims to evaluate the critique and correction capabilities of VLMs, which are two essential building blocks towards VLM self-improvement. VISCO requires VLMs to critique the correctness of each step in CoT, provide natural language explanation, and corrects the CoT based on the critique. VISCO 旨在评估 VLM 的评判（critique）和纠正（correction）能力，这两个能力是 VLM 自主提升推理性能的基础。对于一个视觉推理问题，给定模型生成的 CoT，VISCO 评测集要求 VLM 评判 CoT 中每个步骤的正确性，提供自然语言解释，并根据评判结果对 CoT 进行修正。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VISCO"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "VISCO", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1215, "released": "2024-12-03", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VISCO"}, {"aliases": [], "benchmark_id": "llm-stats-visfactor", "caveat": "VisFactor is a benchmark evaluating fine-grained visual factor perception and reasoning over images.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VisFactor", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1216, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/visfactor?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vision2web", "caveat": "Vision2Web evaluates multimodal models on converting visual designs and screenshots into functional web pages, measuring end-to-end design-to-code capability.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "Vision2Web", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1217, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vision2web?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1993-vistorybench", "caveat": "ViStoryBench is a benchmark designed to rigorously evaluate the capabilities of multimodal generative models (e.g., diffusion models, LLM-based agents) in synthesizing visually coherent image sequences from textual narratives and reference images. ViStoryBench 是一个面向故事可视化任务的综合性评测基准，旨在评估多模态生成模型（如扩散模型视频生成模型等）根据给定叙事文本和参考图像生成视觉连贯且情节一致的图像序列的能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ViStoryBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ViStoryBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1218, "released": "2025-06-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ViStoryBench"}, {"aliases": [], "benchmark_id": "opencompass-1748-visualpuzzles", "caveat": "VisualPuzzles is a benchmark that targets visual reasoning while deliberately minimizing reliance on specialized knowledge. VisualPuzzles consists of 1168 diverse questions spanning five categories: algorithmic, analogical, deductive, inductive, and spatial reasoning. LLM 能考公务员吗？我们做了个测试…\n\n近年来，大模型（LLM）的能力突飞猛进，似乎“越来越聪明”了。但有一个关键问题仍然摆在眼前：\n🤔 它们真的会“推理”吗？\n\n🚀 我们设计了一个名为 VisualPuzzles 🧩 的全新数据集，专门用来回答这个问题：\n脱离专业知识的支持，大模型能靠逻辑本身解题吗？\n我们从多个来源精心挑选或改编了 1168 道图文逻辑题，其中一个重要来源便是中国国家公务员考试行测中的逻辑推理题（没错，真·考公难度）🎯", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VisualPuzzles"], "document_share": 0.0008271298593879239, "domain": "强推理", "name": "VisualPuzzles", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1219, "released": "2025-04-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VisualPuzzles"}, {"aliases": [], "benchmark_id": "opencompass-1634-visualsimpleqa", "caveat": "VisualSimpleQA is a multimodal fact-seeking benchmark with two key features. VisualSimpleQA 是一个多模态事实寻求基准，具有两个关键特性。首先，它使视觉和语言模态中 LVLMs 的评估更加简化和解耦。其次，它纳入了明确的难度标准，以指导人工标注并促进提取具有挑战性的子集，即 VisualSimpleQA-hard。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VisualSimpleQA"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "VisualSimpleQA", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1220, "released": "2025-03-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VisualSimpleQA"}, {"aliases": [], "benchmark_id": "llm-stats-visualwebbench", "caveat": "A multimodal benchmark designed to assess the capabilities of multimodal large language models (MLLMs) across web page understanding and grounding tasks. Comprises 7 tasks (captioning, webpage QA, heading OCR, element OCR, element grounding, action prediction, and action grounding) with 1.5K human-curated instances from 139 real websites across 87 sub-domains.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VisualWebBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1221, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/visualwebbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-visulogic", "caveat": "VisuLogic evaluates logical reasoning capabilities in visual contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VisuLogic", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1222, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/visulogic?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vita-bench", "caveat": "VITA-Bench evaluates AI agents on real-world virtual task automation, measuring their ability to complete complex multi-step tasks in simulated environments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "VITA-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1223, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vita-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2388-vknowu", "caveat": "尽管多模态大型语言模型（MLLMs）在识别物体方面已相当熟练，但它们往往缺乏对世界潜在物理和社会原则的直觉的、类似人类的理解。这种高级的、基于视觉的语义，我们称之为视觉知识，其构成了感知与推理之间的桥梁。为了系统地评估这种能力，我们提出了 VKnowU，一个包含 1,249 个视频、1,680 个问题的综合基准，涵盖了 8 种核心类型的视觉知识，既包括以世界为中心的（例如直觉物理），也包括以人类为中心的任务（例如主观意图）。 尽管多模态大型语言模型（MLLMs）在识别物体方面已相当熟练，但它们往往缺乏对世界潜在物理和社会原则的直觉的、类似人类的理解。这种高级的、基于视觉的语义，我们称之为视觉知识，其构成了感知与推理之间的桥梁。为了系统地评估这种能力，我们提出了 VKnowU，一个包含 1,249 个视频、1,680 个问题的综合基准，涵盖了 8 种核心类型的视觉知识，既包括以世界为中心的（例如直觉物理），也包括以人类为中心的任务（例如主观意图）。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VKnowU"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "VKnowU", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1224, "released": "2025-11-25", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VKnowU"}, {"aliases": [], "benchmark_id": "llm-stats-vladbench", "caveat": "VLADBench is a vision-language autonomous-driving benchmark evaluating understanding of dynamic traffic scenes and participants.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VLADBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1225, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vladbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1541-vlm2-bench", "caveat": "VLM²-Bench is the first comprehensive benchmark that evaluates vision-language models' (VLMs) ability to visually link matching cues across multi-image sequences and videos. The benchmark consists of 9 subtasks with over 3,000 test cases. VLM²-Bench 是第一个全面评估视觉语言模型（VLMs）在多图像序列和视频中视觉链接匹配线索能力的基准。该基准包括 9 个子任务，超过 3000 个测试案例，旨在评估人类日常使用的根本视觉链接能力。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VLM2-Bench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "VLM2-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1226, "released": "2025-02-17", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VLM2-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-vlmsarebiased", "caveat": "VLMsAreBiased evaluates whether vision-language models rely on visual evidence or fall back on language priors when answering.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VLMsAreBiased", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1227, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsarebiased?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vlmsareblind", "caveat": "A vision-language benchmark that probes blind spots and brittle reasoning in multimodal models.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VLMsAreBlind", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1228, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vlmsareblind?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vocalsound", "caveat": "A dataset for improving human vocal sounds recognition, containing over 21,000 crowdsourced recordings of laughter, sighs, coughs, throat clearing, sneezes, and sniffs from 3,365 unique subjects. Used for audio event classification and recognition of human non-speech vocalizations.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500"], "document_share": 0.0008271298593879239, "domain": "audio", "name": "VocalSound", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1229, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vocalsound?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-voicebench-avg", "caveat": "VoiceBench is the first benchmark designed to provide a multi-faceted evaluation of LLM-based voice assistants, evaluating capabilities including general knowledge, instruction-following, reasoning, and safety using both synthetic and real spoken instruction data with diverse speaker characteristics and environmental conditions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "VoiceBench Avg", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1230, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/voicebench-avg?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vqa-rad", "caveat": "VQA-RAD (Visual Question Answering in Radiology) is the first manually constructed dataset of medical visual question answering containing 3,515 clinically generated visual questions and answers about radiology images. The dataset includes questions created by clinical trainees on 315 radiology images from MedPix covering head, chest, and abdominal scans, designed to support AI development for medical image analysis and improve patient care.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VQA-Rad", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1231, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqa-rad?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vqav2", "caveat": "VQAv2 is a balanced Visual Question Answering dataset that addresses language bias by providing complementary images for each question, forcing models to rely on visual understanding rather than language priors. It contains approximately twice the number of image-question pairs compared to the original VQA dataset.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VQAv2", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1232, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vqav2-test", "caveat": "VQA v2.0 (Visual Question Answering v2.0) is a balanced dataset designed to counter language priors in visual question answering. It consists of complementary image pairs where the same question yields different answers, forcing models to rely on visual understanding rather than language bias. The dataset contains 1,105,904 questions across 204,721 COCO images, requiring understanding of vision, language, and commonsense knowledge.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VQAv2 (test)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1233, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28test%29?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-vqav2-val", "caveat": "VQAv2 is a balanced Visual Question Answering dataset containing open-ended questions about images that require understanding of vision, language, and commonsense knowledge to answer. VQAv2 addresses bias issues from the original VQA dataset by collecting complementary images such that every question is associated with similar images that result in different answers, forcing models to actually understand visual content rather than relying on language priors.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "VQAv2 (val)", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1234, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/vqav2-%28val%29?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2383-vrbench", "caveat": "VRBench is the first evaluation benchmark specifically designed to assess multi-step reasoning capabilities over long videos. The benchmark comprises 1,010 long videos with an average duration of 1.6 hours, covering 8 languages and 7 video types. It includes annotations for 9,468 reasoning steps. VRBench是首个专门针对长视频多步推理能力设计的评测基准VRBench。该基准包含 1010 个平均时长1.6小时的长视频，覆盖 8 种语言和 7 种视频类型，标注了 9468 个多步推理问答对和超过 3 万个详细推理步骤。本评测集能够同时对模型的推理过程和推理结果进行评分，是首个同时具备多步推理链标注和评测能力的视频评测集。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/VRBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "VRBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1235, "released": "2025-06-12", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/VRBench"}, {"aliases": [], "benchmark_id": "llm-stats-we-math", "caveat": "We-Math evaluates multimodal models on visual mathematical reasoning, requiring models to understand and solve math problems presented with visual elements such as diagrams, charts, and geometric figures.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500"], "document_share": 0.0008271298593879239, "domain": "math", "name": "We-Math", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1236, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/we-math?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-web-bench", "caveat": "Web Bench evaluates agents on realistic web-development engineering tasks, measuring end-to-end implementation in browser-based workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "Web Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1237, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/web-bench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-webarena-verified", "caveat": "WebArena-Verified evaluates browser agents on realistic web tasks using a verified task set and execution-based grading.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "WebArena-Verified", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1238, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/webarena-verified?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-webdev-arena", "caveat": "WebDev Arena is a leaderboard for evaluating AI models on web development tasks, including zero-shot generation, complex prompts, and interactive web UI creation. Models are ranked using Elo ratings based on their performance in coding and web development challenges.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "WebDev Arena", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1239, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/webdev-arena?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1982-webui-bench", "caveat": "WebUIBench is a comprehensive benchmark designed to evaluate Multimodal Large Language Models (MLLMs) in Web UI-to-code generation tasks across four key capabilities: UI perception, HTML programming, UI-code understanding, and end-to-end transformation. WebUIBench 是一个面向多模态大语言模型（MLLMs）的综合性评测基准，旨在系统评估模型在 Web UI 到代码生成任务中的四个关键能力：界面感知、HTML 编程、界面-代码理解以及整体转换能力。该基准包含来自 700 多个真实网站的 21,000 个高质量问答对，支持对模型在各阶段的细粒度能力分析。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WebUI-Bench"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "WebUI-Bench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1240, "released": "2025-06-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WebUI-Bench"}, {"aliases": [], "benchmark_id": "llm-stats-webvoyager", "caveat": "WebVoyager evaluates an agent's ability to navigate and complete tasks on real websites by perceiving page screenshots and executing browser actions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "WebVoyager", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1241, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/webvoyager?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1279-whodunitbench", "caveat": "WhodunitBench is used to evaluate large multimodal agent under complex tasks and dynamic scenarios. WhodunitBench用于评估大型多模式代理在复杂任务场景下的动态评估。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WhodunitBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "WhodunitBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1242, "released": "2024-09-26", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WhodunitBench"}, {"aliases": [], "benchmark_id": "opencompass-504-wic", "caveat": "WiC is a benchmark for the evaluation of context-sensitive word embeddings. WiC is framed as a binary classification task. Each instance in WiC has a target word w, either a verb or a noun, for which two contexts are provided. Each of these contexts triggers a specific meaning of w. The task is to identify if the occurrences of w in the two contexts correspond to the same meaning or not. In fact, the dataset can also be viewed as an application of Word Sense Disambiguation in practise. Word-in-Context是一个词义消歧任务，被视为句子对的二元分类。给定两个文本片段和一个在两个句子中都出现的多义词，任务是确定该词在两个句子中是否具有相同的含义。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WiC"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "WiC", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1243, "released": "2018-08-28", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WiC"}, {"aliases": [], "benchmark_id": "llm-stats-widesearch", "caveat": "WideSearch is an agentic search benchmark that evaluates models' ability to perform broad, parallel search operations across multiple sources. It tests wide-coverage information retrieval and synthesis capabilities.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "WideSearch", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1244, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/widesearch?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1318-wikicontradict", "caveat": "WikiContradict is a benchmark consisting of 253 high-quality, human-annotated instances designed to assess LLM performance when augmented with retrieved passages containing real-world knowledge conflicts. WikiContradict旨在评估LLM遇到包含真实世界知识冲突的段落检索增强时的性能，由253个高质量的人工注释实例组成。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WikiContradict"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "WikiContradict", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1245, "released": "2024-06-19", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WikiContradict"}, {"aliases": [], "benchmark_id": "opencompass-1129-wikisql", "caveat": "WikiSQL is a dataset of 80654 hand-annotated examples of questions and SQL queries distributed across 24241 tables from Wikipedia that is an order of magnitude larger than comparable datasets. WikiSQL 是一个包含 80,654 个手动标注示例的问题和 SQL 查询的数据集，分布在来自维基百科的 24,241 个表格中。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WikiSQL"], "document_share": 0.0008271298593879239, "domain": "其他", "name": "WikiSQL", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1246, "released": "2017-11-09", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WikiSQL"}, {"aliases": [], "benchmark_id": "llm-stats-wild-bench", "caveat": "WildBench is an automated evaluation framework that benchmarks large language models using 1,024 challenging, real-world tasks selected from over one million human-chatbot conversation logs. It introduces two evaluation metrics (WB-Reward and WB-Score) that achieve high correlation with human preferences and uses task-specific checklists for systematic evaluation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Wild Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1247, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/wild-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1555-wildbench", "caveat": "Weintroduce WildBench, an automated evaluation framework designed to bench-mark large language models (LLMs) using challenging, real-world user queries, which consists of 1,024 tasks carefully selected from over one million human-chatbot conversation logs. WildBench推出自动评估框架和数据集，基于真实用户难题评测大语言模型，包含从逾百万人机对话日志中精选的1,024个任务样本。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WildBench"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "WildBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1248, "released": "2024-06-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WildBench"}, {"aliases": [], "benchmark_id": "llm-stats-wildclawbench", "caveat": "WildClawBench is an agentic coding benchmark from InternLM/Claw-Eval that reports overall model performance on real-world tool-using development tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "WildClawBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1249, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/wildclawbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-2445-wildclawbench", "caveat": "WildClawBench evaluates AI agents across six task categories: productivity flow, code intelligence, social interaction, search & retrieval, creative synthesis, and safety alignment. It comprises 60 hand-authored tasks inside a live environment with real tools — browser, bash, file system, etc. WildClawBench 从六个任务维度评测 AI 智能体：生产力流程、代码智能、社交互动、搜索与检索、创意合成和安全对齐。基准包含 60 个人工设计的任务，运行在配备真实工具（浏览器、bash、文件系统、电子邮件等）的真实环境中。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WildClawBench"], "document_share": 0.0008271298593879239, "domain": "智能体", "name": "WildClawBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1250, "released": "2026-04-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WildClawBench"}, {"aliases": [], "benchmark_id": "llm-stats-winogrande", "caveat": "WinoGrande: An Adversarial Winograd Schema Challenge at Scale. A large-scale dataset of 44,000 pronoun resolution problems designed to test machine commonsense reasoning. Uses adversarial filtering to reduce spurious biases and provides a more robust evaluation of whether AI systems truly understand commonsense or exploit statistical shortcuts. Current best AI methods achieve 59.4-79.1% accuracy, significantly below human performance of 94.0%.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Winogrande", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1251, "released": "2019-11-21", "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/winogrande?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1109-winogrande", "caveat": "WINOGRANDE is a large-scale dataset of 44k problems, inspired by the original WSC design, but adjusted to improve both the scale and the hardness of the dataset. WINOGRANDE 包含 44,000 个问题，受到 WSC 设计的启发，但进行了调整，以提高数据集的规模和难度。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WinoGrande"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "WinoGrande", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1252, "released": "2019-11-21", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WinoGrande"}, {"aliases": [], "benchmark_id": "llm-stats-wmdp", "caveat": "Weapons of Mass Destruction (WMDP) is a multiple-choice benchmark on dual-use biology, chemistry, and cyber knowledge. It measures a model's capacity to enable malicious actors to design, synthesize, acquire, or use chemical, biological, radiological, or nuclear (CBRN) weapons.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "WMDP", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1253, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/wmdp?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-wmt23", "caveat": "The Eighth Conference on Machine Translation (WMT23) benchmark evaluating machine translation systems across 8 language pairs (14 translation directions) including general, biomedical, literary, and low-resource language translation tasks. Features specialized shared tasks for quality estimation, metrics evaluation, sign language translation, and discourse-level literary translation with professional human assessment.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "WMT23", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1254, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt23?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-wmt24", "caveat": "WMT24++ is a comprehensive multilingual machine translation benchmark that expands the WMT24 dataset to cover 55 languages and dialects. It includes human-written references and post-edits across four domains (literary, news, social, and speech) to evaluate machine translation systems and large language models across diverse linguistic contexts.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500"], "document_share": 0.0008271298593879239, "domain": "language", "name": "WMT24++", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1255, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/wmt24%2B%2B?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-workspace-bench", "caveat": "Workspace Bench evaluates AI agents on high-economic-value workplace tasks that span multi-step planning, file processing, and tool use across realistic office and productivity workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "Workspace Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1256, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/workspace-bench?top_n=500"}, {"aliases": ["Workspace-Bench", "Workspace Bench", "Workspace-Bench 1.0"], "benchmark_id": "workspacebench", "caveat": "Large-scale file-workspace tasks; the environment and file dependencies are part of the measurement.", "document_count": 1, "document_ids": ["model_reports:tencent_hy4_preview"], "document_share": 0.0008271298593879239, "domain": "professional", "name": "WorkspaceBench", "organization_count": 1, "organizations": ["Tencent"], "rank": 1257, "released": "2026-05-05", "source": "model_reports", "url": "https://arxiv.org/abs/2605.03596"}, {"aliases": [], "benchmark_id": "llm-stats-worldbench", "caveat": "WorldBench evaluates real-world visual knowledge and understanding across diverse everyday scenes.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "WorldBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1258, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/worldbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1938-worldgenbench", "caveat": null, "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WorldGenBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "WorldGenBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1259, "released": "2025-06-16", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WorldGenBench"}, {"aliases": [], "benchmark_id": "opencompass-1735-worldscore", "caveat": "WorldScore benchmark is the first unified benchmark for world generation. WorldScore基准测试，这是首个用于世界生成的统一基准测试。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WorldScore"], "document_share": 0.0008271298593879239, "domain": "创作", "name": "WorldScore", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1260, "released": "2025-04-01", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WorldScore"}, {"aliases": [], "benchmark_id": "llm-stats-worldvqa", "caveat": "WorldVQA is a benchmark designed to evaluate atomic vision-centric world knowledge. It assesses models' ability to understand and reason about visual elements representing real-world knowledge.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "WorldVQA", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1261, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/worldvqa?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-writingbench", "caveat": "A comprehensive benchmark for evaluating large language models' generative writing capabilities across 6 core writing domains (Academic & Engineering, Finance & Business, Politics & Law, Literature & Art, Education, Advertising & Marketing) and 100 subdomains. Contains 1,239 queries with a query-dependent evaluation framework that dynamically generates 5 instance-specific assessment criteria for each writing task, using a fine-tuned critic model to score responses on style, format, and length dimensions.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "legal", "name": "WritingBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1262, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/writingbench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1701-writingbench", "caveat": "WritingBench: A Comprehensive Benchmark for Generative Writing WritingBench: A Comprehensive Benchmark for Generative Writing", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WritingBench"], "document_share": 0.0008271298593879239, "domain": "长文本", "name": "WritingBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1263, "released": "2025-03-07", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WritingBench"}, {"aliases": [], "benchmark_id": "opencompass-507-wsc", "caveat": "WSC is a pronoun disambiguation task, which requires to determine which noun the pronoun refers to according to the context. WSC是一个代词消歧任务，要求根据上下文判断代词指代的是哪个名词。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/WSC"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "WSC", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1264, "released": null, "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/WSC"}, {"aliases": [], "benchmark_id": "opencompass-1072-xcodeeval", "caveat": "xCodeEval is the largest executable multilingual multitask benchmark to date consisting of 25M document-level coding examples (16.5B tokens) from about 7.5K unique problems covering up to 11 programming languages with execution-level parallelism. xCodeEval 是迄今为止最大的可执行多语言多任务基准，包含 2500 万个文档级编码示例（165 亿个标记），来自约 7500 个独特问题，涵盖多达 11 种编程语言。它包括 7 个任务，涉及代码理解、生成、翻译和检索。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/xCodeEval"], "document_share": 0.0008271298593879239, "domain": "代码", "name": "xCodeEval", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1265, "released": "2023-11-06", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/xCodeEval"}, {"aliases": [], "benchmark_id": "llm-stats-xdailybench", "caveat": "xDailyBench evaluates AI agents on white-collar office work, covering everyday professional tasks such as document handling, consultation, and multi-step productivity workflows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "xDailyBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1266, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/xdailybench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-xlsum-english", "caveat": "Large-scale multilingual abstractive summarization dataset comprising 1 million professionally annotated article-summary pairs from BBC, covering 44 languages. XL-Sum is highly abstractive, concise, and of high quality, designed to encourage research on multilingual abstractive summarization tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500"], "document_share": 0.0008271298593879239, "domain": "summarization", "name": "XLSum English", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1267, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/xlsum-english?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-xstest", "caveat": "XSTest is a test suite designed to identify exaggerated safety behaviours in large language models. It comprises 450 prompts: 250 safe prompts across ten prompt types that well-calibrated models should not refuse to comply with, and 200 unsafe prompts as contrasts that models should refuse. The benchmark systematically evaluates whether models refuse to respond to clearly safe prompts due to overly cautious safety mechanisms.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500"], "document_share": 0.0008271298593879239, "domain": "safety", "name": "XSTest", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1268, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/xstest?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-521-xsum", "caveat": "XSum is a single-document summarization task which does not favor extractive strategies and calls for an abstractive modeling approach. The idea is to create a short, one-sentence news summary answering the question “What is the article about?”. The dataset collects real-world, large scale data by harvesting online articles from the British Broadcasting Corporation (BBC). XSum是一个单文档摘要任务，不支持抽取式策略，需要采用抽象建模方法。其思想是创建一个简短的一句话新闻摘要，回答“这篇文章是关于什么的？”的问题。该数据集通过从英国广播公司（BBC）收集在线文章，得到了大量的现实数据。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/XSum"], "document_share": 0.0008271298593879239, "domain": "理解", "name": "XSum", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1269, "released": "2018-08-27", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/XSum"}, {"aliases": [], "benchmark_id": "opencompass-1784-xverify", "caveat": "xVerify, an efficient answer verifier for reasoning model evaluations. xVerify demonstrates strong capability in equivalence judgment, enabling it to effectively determine whether the answers produced by reasoning models are equivalent to reference answers across various types of objective questions xVerify，这是一种用于推理模型评估的高效答案验证器。xVerify 在等价判断方面表现出强大的能力，使其能够有效地确定推理模型生成的答案是否等同于各种类型客观问题的参考答案。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/xVerify"], "document_share": 0.0008271298593879239, "domain": "推理", "name": "xVerify", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1270, "released": "2025-04-14", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/xVerify"}, {"aliases": [], "benchmark_id": "llm-stats-yc-bench", "caveat": "YC-Bench evaluates agents on long-horizon, open-ended business and investment decision-making. The reported metric is the final assets (fund value, in US dollars) accumulated by the agent over the course of the simulation.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "finance", "name": "YC-Bench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1271, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/yc-bench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1017-yue-benchmark", "caveat": "The benchmarks introduced for evaluating large language models (LLMs) on Cantonese include Yue-TruthfulQA, Yue-GSM8K, Yue-ARC-C, Yue-MMLU, and Yue-TRANS. Each of these benchmarks focuses on different aspects of language understanding and generation in Cantonese, offering a comprehensive means of assessing the capabilities of LLMs in handling this language. Yue_Benchmark用于评估粤语大型语言模型（LLMs）。该评测集包含：Yue TruthtyQA、Yue-GSM8K、Yue-ARC-C、Yue MMLU和Yue TRANS，侧重于粤语语言理解和生成的不同方面，为评估LLM粤语能力提供了一种全面的方法。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/Yue_Benchmark"], "document_share": 0.0008271298593879239, "domain": "语言", "name": "Yue_Benchmark", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1272, "released": "2024-08-31", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/Yue_Benchmark"}, {"aliases": [], "benchmark_id": "llm-stats-zclawbench", "caveat": "ZClawBench evaluates Claw-style agent task execution quality, measuring a model's ability to autonomously complete complex multi-step coding tasks in real-world environments.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "agents", "name": "ZClawBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1273, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/zclawbench?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-zebralogic", "caveat": "ZebraLogic is an evaluation framework for assessing large language models' logical reasoning capabilities through logic grid puzzles derived from constraint satisfaction problems (CSPs). The benchmark consists of 1,000 programmatically generated puzzles with controllable and quantifiable complexity, revealing a 'curse of complexity' where model accuracy declines significantly as problem complexity grows.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500"], "document_share": 0.0008271298593879239, "domain": "reasoning", "name": "ZebraLogic", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1274, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/zebralogic?top_n=500"}, {"aliases": [], "benchmark_id": "llm-stats-zerobench", "caveat": "ZEROBench is a challenging vision benchmark designed to test models on zero-shot visual understanding tasks.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ZEROBench", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1275, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench?top_n=500"}, {"aliases": [], "benchmark_id": "opencompass-1532-zerobench", "caveat": "ZeroBench is a challenging visual reasoning benchmark for LMMs. It consists of a main set of 100 high-quality, manually curated questions covering numerous domains, reasoning types and image type. Questions have been designed and calibrated to be beyond the capabilities of current frontier models. ZeroBench 是针对多模态模型（LMMs）的具有挑战性的视觉推理基准。它由一组主要的 100 个高质量人工问题组成，涵盖多个领域、推理类型和图像类型。ZeroBench 中的问题经过设计和校准，已经超出了当前前沿模型的能力范围。", "document_count": 1, "document_ids": ["opencompass_hub:https://hub.opencompass.org.cn/dataset-detail/ZeroBench"], "document_share": 0.0008271298593879239, "domain": "多模态", "name": "ZeroBench", "organization_count": 1, "organizations": ["opencompass_hub"], "rank": 1276, "released": "2025-02-13", "source": "opencompass_hub", "url": "https://hub.opencompass.org.cn/dataset-detail/ZeroBench"}, {"aliases": [], "benchmark_id": "llm-stats-zerobench-sub", "caveat": "ZEROBench-Sub is a subset of the ZEROBench benchmark.", "document_count": 1, "document_ids": ["llm_stats:https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500"], "document_share": 0.0008271298593879239, "domain": "multimodal", "name": "ZEROBench-Sub", "organization_count": 1, "organizations": ["llm_stats"], "rank": 1277, "released": null, "source": "llm_stats", "url": "https://api.zeroeval.com/leaderboard/benchmarks/zerobench-sub?top_n=500"}, {"aliases": [], "benchmark_id": "artificial-analysis-tau2-bench-telecom", "caveat": "Agentic tool use", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/tau2-bench"], "document_share": 0.0008271298593879239, "domain": "agentic", "name": "τ²-Bench Telecom", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 1278, "released": "2025-06-12", "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/tau2-bench"}, {"aliases": [], "benchmark_id": "artificial-analysis-tau3-banking", "caveat": "Agentic tool use", "document_count": 1, "document_ids": ["artificial_analysis:https://artificialanalysis.ai/evaluations/tau3-banking"], "document_share": 0.0008271298593879239, "domain": "intelligence-index", "name": "τ³-Banking", "organization_count": 1, "organizations": ["artificial_analysis"], "rank": 1279, "released": null, "source": "artificial_analysis", "url": "https://artificialanalysis.ai/evaluations/tau3-banking"}, {"aliases": ["ASI-Bench", "ASI Bench"], "benchmark_id": "asi_bench", "caveat": "First benchmark that jointly evaluates general intelligence, innovation, and autonomous execution, via 60 project-level scientific research tasks spanning 11 domains, progressively withdrawing human guidance. Very new (arXiv 2608.17271, code at github.com/apexin-ai/ASI-Bench) and intended to probe superintelligence-level behaviour, so high variance across seeds and scaffolds should be expected before any vendor adopts it. Only the seed-31415 instances are public; the seed-42 references stay private for official leaderboard scoring, so a locally computed score and a leaderboard score are not the same measurement. This registry's maintainer is one of the benchmark's authors.", "document_count": 0, "document_ids": [], "document_share": 0.0, "domain": "science", "name": "ASI-Bench", "organization_count": 0, "organizations": [], "rank": 1280, "released": "2026-08-18", "source": "model_reports", "url": "https://arxiv.org/abs/2608.17271"}, {"aliases": ["JointAVBench", "Joint AV-Bench"], "benchmark_id": "jointavbench", "caveat": "An ICLR 2026 poster on joint audio-visual reasoning in Omni-LLMs, built with strict audio-video correlation across five cognitive dimensions and four audio information types. Because the question set requires both modalities, vendors that only report vision-only or audio-only figures are not comparable on this instrument. Automated annotation pipeline; manual verification effort is not quantified.", "document_count": 0, "document_ids": [], "document_share": 0.0, "domain": "multimodal", "name": "JointAVBench", "organization_count": 0, "organizations": [], "rank": 1281, "released": "2025-12-14", "source": "model_reports", "url": "https://arxiv.org/abs/2512.12772"}, {"aliases": ["OmniVideoBench", "Omni Video-Bench"], "benchmark_id": "omnivideobench", "caveat": "1,000 audio-visual reasoning questions over 628 videos (seconds to 30 minutes long), targeting synergistic audio-visual understanding in Omni MLLMs, by the NJU-LINK team; an ICLR 2026 poster. Video length and audio-video correlation vary widely, so per-model results depend heavily on which subset is sampled and on frame/audio chunking strategy.", "document_count": 0, "document_ids": [], "document_share": 0.0, "domain": "multimodal", "name": "OmniVideoBench", "organization_count": 0, "organizations": [], "rank": 1282, "released": "2025-10-12", "source": "model_reports", "url": "https://arxiv.org/abs/2510.10689"}, {"aliases": ["RSI-Bench", "RSI Bench", "RSIBench", "Recursive Self-Improvement Benchmark"], "benchmark_id": "rsi_bench", "caveat": "Announced by Scale AI on 2026-08-07 and still in preview: it is soliciting task contributions rather than reporting results, so no model card in this registry reports it and its adoption count is zero. That zero is a real reading, not a gap in the crawl. Recorded now so the benchmark is addressable by name before any score exists, and so the first card to report it has an id to resolve against. Scores are per-task against a baseline under a fixed compute budget, graded on reliability, efficiency, generality and idea quality, so a single headline number should not be expected. Not to be confused with the llm-stats \"RSI Index\" record, which is a different instrument.", "document_count": 0, "document_ids": [], "document_share": 0.0, "domain": "agent", "name": "RSI-Bench", "organization_count": 0, "organizations": [], "rank": 1283, "released": "2026-08-07", "source": "model_reports", "url": "https://www.rsi-benchmark.com/"}, {"aliases": ["SWE-bench Science", "SWE-Bench-Science", "SWE Bench Science"], "benchmark_id": "swe_bench_science", "caveat": "Repository-level scientific software engineering benchmark from the OpenMOSS team (Fudan): 119 tasks across 98 GitHub repositories in 20 scientific domains, organized into Issue-driven, Expert-exploratory, and Engineering-integration paradigms. Best reported pass@1 is below 50% (Claude Code with Opus-5 max), and the paper identifies four recurring failure mechanisms, so scores are highly sensitive to scientific-domain knowledge and tool access, not only to coding ability.", "document_count": 0, "document_ids": [], "document_share": 0.0, "domain": "coding_agent", "name": "SWE-bench Science", "organization_count": 0, "organizations": [], "rank": 1284, "released": "2026-08-20", "source": "model_reports", "url": "https://arxiv.org/abs/2608.19799"}], "measures": "Counts distinct source documents that record each benchmark. Model reports and registry pages use the same rule. Each document counts once per benchmark record. This measures documentation coverage, not benchmark quality.", "organization_count": 16, "organizations": {"Ai2": 1, "Anthropic": 4, "ApodexAI": 1, "DeepSeek": 5, "Google": 4, "Meta": 2, "Mistral": 3, "Moonshot AI": 1, "OpenAI": 5, "Qwen": 4, "Tencent": 1, "Z.ai": 5, "artificial_analysis": 23, "llm_stats": 687, "opencompass_hub": 461, "xAI": 2}, "schema_version": 1}, "schema_version": 1}
